From ea7e0426eee2a39aa8eff721686fc0e7aa6a5c2f Mon Sep 17 00:00:00 2001 From: leemr Date: Fri, 25 Sep 2026 14:38:37 -0400 Subject: [PATCH 1/4] feat: support JSON5 leading and trailing decimal points, Infinity and NaN Numbers like `.5`, `5.` and `5.e3` now parse, and so do `Infinity`, `-Infinity`, `+Infinity`, `NaN` and `-NaN`. Two new `ParseOptions` switches gate them: `allow_bare_decimal_point_numbers` and `allow_non_finite_numbers`. Both default to `true`, like the other JSON5 switches. The `serde_json` conversion turns `.5` into `0.5` and `5.` into the float `5.0`. `serde_json` cannot hold `Infinity` or `NaN`, so the deserializer reports them as out of range, and the AST and CST conversions keep their fallback to a string. A property name like `Infinity_count` or `NaN-key` still scans as one word, and `[true-1]` still parses as `[true, -1]` when commas can be missing. --- src/ast.rs | 20 ++++++++- src/common.rs | 18 ++++++++ src/cst/mod.rs | 13 +++++- src/errors.rs | 8 ++++ src/lib.rs | 2 + src/parse_to_ast.rs | 53 ++++++++++++++++++++++ src/parse_to_value.rs | 27 ++++++++++++ src/parser.rs | 2 + src/scanner.rs | 100 +++++++++++++++++++++++++++++++++++++++++- src/serde.rs | 19 ++++++++ 10 files changed, 256 insertions(+), 6 deletions(-) diff --git a/src/ast.rs b/src/ast.rs index 21767c7..b401684 100644 --- a/src/ast.rs +++ b/src/ast.rs @@ -87,8 +87,8 @@ impl<'a> From> for serde_json::Value { } } else { // standard decimal number - let num_for_parsing = num.value.trim_start_matches('+'); - match serde_json::Number::from_str(num_for_parsing) { + let num_for_parsing = crate::common::fill_bare_decimal_point(num.value.trim_start_matches('+')); + match serde_json::Number::from_str(&num_for_parsing) { Ok(number) => serde_json::Value::Number(number), Err(_) => serde_json::Value::String(num.value.to_string()), } @@ -656,6 +656,22 @@ mod test { ); } + #[cfg(feature = "serde_json")] + #[test] + fn it_should_coerce_json5_numbers_to_serde_value() { + let ast = parse_to_ast( + r#"[.5, -.5, +5., 5.e3, Infinity]"#, + &Default::default(), + &ParseOptions::default(), + ) + .unwrap(); + let serde_value: serde_json::Value = ast.value.unwrap().into(); + + // serde_json cannot hold Infinity, so it keeps the existing fallback to a string + assert_eq!(serde_value, serde_json::json!([0.5, -0.5, 5.0, 5000.0, "Infinity"])); + assert!(serde_value[2].is_f64()); + } + #[cfg(feature = "serde_json")] #[test] fn handle_weird_data() { diff --git a/src/common.rs b/src/common.rs index 9f1ad44..cd5c3b1 100644 --- a/src/common.rs +++ b/src/common.rs @@ -23,6 +23,24 @@ impl Ranged for Range { } } +/// Adds the digit serde_json needs beside a leading or trailing decimal point (ex. `.5` to `0.5`). +#[cfg(feature = "serde_json")] +pub(crate) fn fill_bare_decimal_point(num: &str) -> std::borrow::Cow<'_, str> { + let Some(dot) = num.find('.') else { + return std::borrow::Cow::Borrowed(num); + }; + let before = num[..dot].ends_with(|c: char| c.is_ascii_digit()); + let after = num[dot + 1..].starts_with(|c: char| c.is_ascii_digit()); + if before && after { + return std::borrow::Cow::Borrowed(num); + } + let mut filled = String::with_capacity(num.len() + 1); + filled.push_str(&num[..dot]); + filled.push_str(if before { ".0" } else { "0." }); + filled.push_str(&num[dot + 1..]); + std::borrow::Cow::Owned(filled) +} + /// Represents an object that has a range in the text. pub trait Ranged { /// Gets the range. diff --git a/src/cst/mod.rs b/src/cst/mod.rs index 3dcd728..0c16644 100644 --- a/src/cst/mod.rs +++ b/src/cst/mod.rs @@ -1518,8 +1518,8 @@ impl CstNumberLit { } } else { // standard decimal number - strip leading + if present (serde_json doesn't accept it) - let num_for_parsing = raw.trim_start_matches('+'); - match serde_json::Number::from_str(num_for_parsing) { + let parsed = serde_json::Number::from_str(&crate::common::fill_bare_decimal_point(raw.trim_start_matches('+'))); + match parsed { Ok(number) => Some(serde_json::Value::Number(number)), // if the number is invalid, return it as a string (same behavior as AST conversion) Err(_) => Some(serde_json::Value::String(raw)), @@ -5203,6 +5203,15 @@ value3: true assert_eq!(value, SerdeValue::Null); } + #[test] + fn test_cst_to_serde_value_json5_numbers() { + let root = build_cst(r#"[.5, -.5, +5., 5.e3, NaN]"#); + let value = root.to_serde_value().unwrap(); + assert_eq!(value, serde_json::json!([0.5, -0.5, 5.0, 5000.0, "NaN"])); + assert!(value[2].is_f64()); + assert_eq!(root.to_string(), "[.5, -.5, +5., 5.e3, NaN]"); + } + #[test] fn test_cst_to_serde_value_array() { let root = build_cst(r#"[1, 2, 3]"#); diff --git a/src/errors.rs b/src/errors.rs index 72cd0fe..d921dae 100644 --- a/src/errors.rs +++ b/src/errors.rs @@ -6,6 +6,7 @@ use super::common::Range; #[derive(Debug)] pub enum ParseErrorKind { + BareDecimalPointNumbersNotAllowed, CommentsNotAllowed, ExpectedColonAfterObjectKey, ExpectedObjectValue, @@ -16,6 +17,7 @@ pub enum ParseErrorKind { HexadecimalNumbersNotAllowed, ExpectedComma, MultipleRootJsonValues, + NonFiniteNumbersNotAllowed, SingleQuotedStringsNotAllowed, String(ParseStringErrorKind), TrailingCommasNotAllowed, @@ -40,6 +42,9 @@ impl std::fmt::Display for ParseErrorKind { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { use ParseErrorKind::*; match self { + BareDecimalPointNumbersNotAllowed => { + write!(f, "Leading or trailing decimal points on numbers are not allowed") + } CommentsNotAllowed => { write!(f, "Comments are not allowed") } @@ -70,6 +75,9 @@ impl std::fmt::Display for ParseErrorKind { MultipleRootJsonValues => { write!(f, "Text cannot contain more than one JSON value") } + NonFiniteNumbersNotAllowed => { + write!(f, "Infinity and NaN are not allowed") + } SingleQuotedStringsNotAllowed => { write!(f, "Single-quoted strings are not allowed") } diff --git a/src/lib.rs b/src/lib.rs index 20c6fd2..1e0dd64 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -115,6 +115,8 @@ //! allow_single_quoted_strings: false, //! allow_hexadecimal_numbers: false, //! allow_unary_plus_numbers: false, +//! allow_bare_decimal_point_numbers: false, +//! allow_non_finite_numbers: false, //! })?; //! # Ok(()) //! # } diff --git a/src/parse_to_ast.rs b/src/parse_to_ast.rs index 3a8d8af..5da13e0 100644 --- a/src/parse_to_ast.rs +++ b/src/parse_to_ast.rs @@ -63,6 +63,10 @@ pub struct ParseOptions { pub allow_hexadecimal_numbers: bool, /// Allow unary plus sign on numbers like +42 (defaults to `true`). pub allow_unary_plus_numbers: bool, + /// Allow a leading or trailing decimal point on numbers like .5 or 5. (defaults to `true`). + pub allow_bare_decimal_point_numbers: bool, + /// Allow the numbers Infinity, -Infinity and NaN (defaults to `true`). + pub allow_non_finite_numbers: bool, } impl Default for ParseOptions { @@ -75,6 +79,8 @@ impl Default for ParseOptions { allow_single_quoted_strings: true, allow_hexadecimal_numbers: true, allow_unary_plus_numbers: true, + allow_bare_decimal_point_numbers: true, + allow_non_finite_numbers: true, } } } @@ -253,6 +259,8 @@ pub fn parse_to_ast<'a>( allow_single_quoted_strings: parse_options.allow_single_quoted_strings, allow_hexadecimal_numbers: parse_options.allow_hexadecimal_numbers, allow_unary_plus_numbers: parse_options.allow_unary_plus_numbers, + allow_bare_decimal_point_numbers: parse_options.allow_bare_decimal_point_numbers, + allow_non_finite_numbers: parse_options.allow_non_finite_numbers, }, ), comments: match collect_options.comments { @@ -591,6 +599,27 @@ mod tests { ); } + #[test] + fn strict_should_error_bare_decimal_point_number() { + assert_has_strict_error( + r#"{ "key": .5 }"#, + "Leading or trailing decimal points on numbers are not allowed on line 1 column 10", + ); + assert_has_strict_error( + r#"{ "key": 5. }"#, + "Leading or trailing decimal points on numbers are not allowed on line 1 column 10", + ); + } + + #[test] + fn strict_should_error_non_finite_number() { + assert_has_strict_error( + r#"{ "key": -Infinity }"#, + "Infinity and NaN are not allowed on line 1 column 10", + ); + assert_has_strict_error(r#"{ "key": NaN }"#, "Unexpected word on line 1 column 10"); + } + #[track_caller] fn assert_has_strict_error(text: &str, message: &str) { let result = parse_to_ast(text, &Default::default(), &strict_options()); @@ -609,6 +638,8 @@ mod tests { allow_single_quoted_strings: false, allow_hexadecimal_numbers: false, allow_unary_plus_numbers: false, + allow_bare_decimal_point_numbers: false, + allow_non_finite_numbers: false, } } @@ -708,6 +739,28 @@ mod tests { assert_eq!(number_value.value, "+42"); } + #[test] + fn it_should_parse_json5_numbers_and_keep_non_finite_words_as_keys() { + let result = parse_to_ast( + r#"{ "a": .5, "b": 5., "c": -Infinity, Infinity: NaN }"#, + &Default::default(), + &Default::default(), + ) + .unwrap(); + + let value = result.value.unwrap(); + let obj = value.as_object().unwrap(); + let values = obj + .properties + .iter() + .map(|p| (p.name.as_str(), p.value.as_number_lit().unwrap().value)) + .collect::>(); + assert_eq!( + values, + vec![("a", ".5"), ("b", "5."), ("c", "-Infinity"), ("Infinity", "NaN")] + ); + } + #[test] fn missing_comma_between_properties() { let text = r#"{ diff --git a/src/parse_to_value.rs b/src/parse_to_value.rs index 774c31c..5720021 100644 --- a/src/parse_to_value.rs +++ b/src/parse_to_value.rs @@ -359,6 +359,8 @@ mod tests { allow_single_quoted_strings: false, allow_hexadecimal_numbers: false, allow_unary_plus_numbers: false, + allow_bare_decimal_point_numbers: false, + allow_non_finite_numbers: false, } } @@ -575,6 +577,31 @@ mod tests { assert_eq!(obj.get_number("CP_CanFuncReqId").unwrap(), "0x7DF"); } + #[test] + fn it_should_parse_loose_property_names_starting_with_infinity_or_nan() { + let value = parse_to_value(r#"{ Infinity_count: 1, NaN-key: 2 }"#, &Default::default()) + .unwrap() + .unwrap(); + let obj = match &value { + JsonValue::Object(o) => o, + _ => panic!("Expected object"), + }; + assert_eq!(obj.get_number("Infinity_count").unwrap(), "1"); + assert_eq!(obj.get_number("NaN-key").unwrap(), "2"); + } + + #[test] + fn it_should_keep_keyword_boundaries_with_missing_commas() { + let value = parse_to_value("[true-1]", &Default::default()).unwrap().unwrap(); + let arr = match value { + JsonValue::Array(a) => a, + _ => panic!("Expected array"), + }; + assert_eq!(arr.len(), 2); + assert!(matches!(arr.get(0), Some(JsonValue::Boolean(true)))); + assert!(matches!(arr.get(1), Some(JsonValue::Number("-1")))); + } + #[test] fn it_should_parse_unary_plus_numbers() { let value = parse_to_value(r#"{ "test": +42 }"#, &Default::default()) diff --git a/src/parser.rs b/src/parser.rs index b7c5101..67ee6e3 100644 --- a/src/parser.rs +++ b/src/parser.rs @@ -47,6 +47,8 @@ impl<'a> JsoncParser<'a> { allow_single_quoted_strings: options.allow_single_quoted_strings, allow_hexadecimal_numbers: options.allow_hexadecimal_numbers, allow_unary_plus_numbers: options.allow_unary_plus_numbers, + allow_bare_decimal_point_numbers: options.allow_bare_decimal_point_numbers, + allow_non_finite_numbers: options.allow_non_finite_numbers, }, ), text, diff --git a/src/scanner.rs b/src/scanner.rs index d99be3d..537389f 100644 --- a/src/scanner.rs +++ b/src/scanner.rs @@ -16,6 +16,8 @@ pub struct Scanner<'a> { allow_single_quoted_strings: bool, allow_hexadecimal_numbers: bool, allow_unary_plus_numbers: bool, + allow_bare_decimal_point_numbers: bool, + allow_non_finite_numbers: bool, } /// Options for the scanner. @@ -27,6 +29,10 @@ pub struct ScannerOptions { pub allow_hexadecimal_numbers: bool, /// Allow unary plus sign on numbers like +42 (defaults to `true`). pub allow_unary_plus_numbers: bool, + /// Allow a leading or trailing decimal point on numbers like .5 or 5. (defaults to `true`). + pub allow_bare_decimal_point_numbers: bool, + /// Allow the numbers Infinity, -Infinity and NaN (defaults to `true`). + pub allow_non_finite_numbers: bool, } impl Default for ScannerOptions { @@ -35,6 +41,8 @@ impl Default for ScannerOptions { allow_single_quoted_strings: true, allow_hexadecimal_numbers: true, allow_unary_plus_numbers: true, + allow_bare_decimal_point_numbers: true, + allow_non_finite_numbers: true, } } } @@ -51,6 +59,8 @@ impl<'a> Scanner<'a> { allow_single_quoted_strings: options.allow_single_quoted_strings, allow_hexadecimal_numbers: options.allow_hexadecimal_numbers, allow_unary_plus_numbers: options.allow_unary_plus_numbers, + allow_bare_decimal_point_numbers: options.allow_bare_decimal_point_numbers, + allow_non_finite_numbers: options.allow_non_finite_numbers, } } @@ -102,6 +112,9 @@ impl<'a> Scanner<'a> { _ => Err(self.create_error_for_current_token(ParseErrorKind::UnexpectedToken)), }, b'-' | b'+' | b'0'..=b'9' => self.parse_number(), + b'.' if matches!(self.bytes.get(self.byte_index + 1), Some(b'0'..=b'9')) => self.parse_number(), + b'I' if self.allow_non_finite_numbers && self.try_move_whole_word("Infinity") => Ok(Token::Number("Infinity")), + b'N' if self.allow_non_finite_numbers && self.try_move_whole_word("NaN") => Ok(Token::Number("NaN")), b't' if self.try_move_word("true") => Ok(Token::Boolean(true)), b'f' if self.try_move_word("false") => Ok(Token::Boolean(false)), b'n' if self.try_move_word("null") => Ok(Token::Null), @@ -248,6 +261,26 @@ impl<'a> Scanner<'a> { self.byte_index += 1; } } + Some(b'.') if matches!(self.bytes.get(self.byte_index + 1), Some(b'0'..=b'9')) => { + if !self.allow_bare_decimal_point_numbers { + return Err(self.create_error_for_current_token(ParseErrorKind::BareDecimalPointNumbersNotAllowed)); + } + } + // only reached after a sign, so unlike in `scan` there is no word to keep whole + Some(b'I' | b'N') => { + let word = if self.bytes[self.byte_index] == b'I' { + "Infinity" + } else { + "NaN" + }; + if !self.try_move_word(word) { + return Err(self.create_error_for_current_char(ParseErrorKind::ExpectedDigitFollowingNegativeSign)); + } + if !self.allow_non_finite_numbers { + return Err(self.create_error_for_current_token(ParseErrorKind::NonFiniteNumbersNotAllowed)); + } + return Ok(Token::Number(&self.file_text[start_byte_index..self.byte_index])); + } _ => { return Err(self.create_error_for_current_char(ParseErrorKind::ExpectedDigitFollowingNegativeSign)); } @@ -256,8 +289,8 @@ impl<'a> Scanner<'a> { if self.bytes.get(self.byte_index) == Some(&b'.') { self.byte_index += 1; - if !matches!(self.bytes.get(self.byte_index), Some(b'0'..=b'9')) { - return Err(self.create_error_for_current_char(ParseErrorKind::ExpectedDigit)); + if !self.allow_bare_decimal_point_numbers && !matches!(self.bytes.get(self.byte_index), Some(b'0'..=b'9')) { + return Err(self.create_error_for_current_token(ParseErrorKind::BareDecimalPointNumbersNotAllowed)); } while matches!(self.bytes.get(self.byte_index), Some(b'0'..=b'9')) { @@ -382,6 +415,15 @@ impl<'a> Scanner<'a> { true } + /// Like `try_move_word`, but also refuses the characters `parse_word` continues a word with, + /// so a property name like `Infinity_count` or `NaN-key` still scans as one word. + fn try_move_whole_word(&mut self, text: &str) -> bool { + if matches!(self.bytes.get(self.byte_index + text.len()), Some(b'-' | b'_')) { + return false; + } + self.try_move_word(text) + } + fn parse_word(&mut self) -> Result, ParseError> { let start_byte_index = self.byte_index; @@ -570,6 +612,60 @@ mod tests { ); } + #[test] + fn it_tokenizes_bare_decimal_point_numbers() { + assert_has_tokens( + ".5, -.5, +.5, 5., -5., 5.e3", + vec![ + Token::Number(".5"), + Token::Comma, + Token::Number("-.5"), + Token::Comma, + Token::Number("+.5"), + Token::Comma, + Token::Number("5."), + Token::Comma, + Token::Number("-5."), + Token::Comma, + Token::Number("5.e3"), + ], + ); + assert_has_error(".", "Unexpected token on line 1 column 1"); + assert_has_error("-.", "Expected digit following negative sign on line 1 column 2"); + } + + #[test] + fn it_tokenizes_non_finite_numbers() { + assert_has_tokens( + "Infinity, -Infinity, +Infinity, NaN, -NaN, Infinityx", + vec![ + Token::Number("Infinity"), + Token::Comma, + Token::Number("-Infinity"), + Token::Comma, + Token::Number("+Infinity"), + Token::Comma, + Token::Number("NaN"), + Token::Comma, + Token::Number("-NaN"), + Token::Comma, + Token::Word("Infinityx"), + ], + ); + assert_has_error("-Infinit", "Expected digit following negative sign on line 1 column 2"); + // with missing commas, a signed keyword ends where `-1-1` would end + assert_has_tokens( + "-Infinity-1, -NaN_", + vec![ + Token::Number("-Infinity"), + Token::Number("-1"), + Token::Comma, + Token::Number("-NaN"), + Token::Word("_"), + ], + ); + } + #[test] fn it_errors_invalid_exponent() { assert_has_error( diff --git a/src/serde.rs b/src/serde.rs index 42cb410..0254b1e 100644 --- a/src/serde.rs +++ b/src/serde.rs @@ -549,6 +549,23 @@ mod tests { assert_eq!(result, SerdeValue::Object(expected_value)); } + #[test] + fn it_should_parse_bare_decimal_point_numbers() { + let result = parse_to_serde_value::(r#"[.5, -.5, 5., 5.e3]"#, &Default::default()).unwrap(); + assert_eq!(result, serde_json::json!([0.5, -0.5, 5.0, 5000.0])); + assert!(result[2].is_f64()); + } + + #[test] + fn it_should_error_for_non_finite_numbers() { + // serde_json has no representation for these + assert_has_error(r#"{ "a": Infinity }"#, "Number is out of range on line 1 column 8"); + assert_has_error("[-Infinity]", "Number is out of range on line 1 column 2"); + assert_has_error("NaN", "Number is out of range on line 1 column 1"); + let value = parse_to_serde_value::("-Infinity", &Default::default()); + assert!(value.is_err()); + } + #[test] fn it_should_error_when_number_is_out_of_f64_range() { assert_has_error(r#"{ "amount": 1e400 }"#, "Number is out of range on line 1 column 13"); @@ -866,6 +883,8 @@ mod tests { allow_single_quoted_strings: false, allow_hexadecimal_numbers: false, allow_unary_plus_numbers: false, + allow_bare_decimal_point_numbers: false, + allow_non_finite_numbers: false, }, ) } From e6c96906242ae681fdf167b263f3cedc262b1482 Mon Sep 17 00:00:00 2001 From: leemr Date: Fri, 25 Sep 2026 14:43:24 -0400 Subject: [PATCH 2/4] feat: support JSON5 string escapes and a lone carriage return Strings now accept the JSON5 escapes: `\'` in a double-quoted string, `\"` in a single-quoted string, `\v`, `\0`, `\xHH`, a backslash before a line terminator (a line continuation), and a backslash before any other character except a digit, which gives that character. A new `ParseOptions` switch, `allow_extended_string_escapes`, gates them and defaults to `true`. With it off, the old errors for `\'` and `\"` stay. A line comment now also ends at a lone `\r`, which is a line terminator in JSON5 and in JavaScript, and error messages count it as a line break. U+2028 and U+2029 do not end a line comment, so this crate keeps agreeing with other JSONC parsers. The CST keeps a lone `\r` as whitespace, so the text round-trips unchanged. --- src/cst/mod.rs | 11 +++++ src/errors.rs | 5 ++- src/lib.rs | 1 + src/parse_to_ast.rs | 5 +++ src/parse_to_value.rs | 1 + src/parser.rs | 1 + src/scanner.rs | 94 ++++++++++++++++++++++++++++++++++++++----- src/serde.rs | 1 + src/string.rs | 69 ++++++++++++++++++++++++++++++- 9 files changed, 175 insertions(+), 13 deletions(-) diff --git a/src/cst/mod.rs b/src/cst/mod.rs index 0c16644..7d3871c 100644 --- a/src/cst/mod.rs +++ b/src/cst/mod.rs @@ -5212,6 +5212,17 @@ value3: true assert_eq!(root.to_string(), "[.5, -.5, +5., 5.e3, NaN]"); } + #[test] + fn test_cst_json5_strings_and_line_terminators() { + let text = "{\r // a\r 'k': 'line\\\ncontinued \\x41\\v',\u{2028} // b\r \"it\\'s\": 1,\r}"; + let root = build_cst(text); + assert_eq!(root.to_string(), text); + assert_eq!( + root.to_serde_value().unwrap(), + serde_json::json!({ "k": "linecontinued A\u{0B}", "it's": 1 }) + ); + } + #[test] fn test_cst_to_serde_value_array() { let root = build_cst(r#"[1, 2, 3]"#); diff --git a/src/errors.rs b/src/errors.rs index d921dae..df34f2d 100644 --- a/src/errors.rs +++ b/src/errors.rs @@ -214,8 +214,9 @@ impl fmt::Display for ParseError { fn get_line_and_column_display(range: Range, file_text: &str) -> (usize, usize) { let mut line_index = 0; let mut column_index = 0; - for c in file_text[..range.start].chars() { - if c == '\n' { + for (i, c) in file_text[..range.start].char_indices() { + // a lone \r ends a line comment, so it also counts as a line break here + if c == '\n' || c == '\r' && file_text.as_bytes().get(i + 1) != Some(&b'\n') { line_index += 1; column_index = 0; } else { diff --git a/src/lib.rs b/src/lib.rs index 1e0dd64..490948e 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -117,6 +117,7 @@ //! allow_unary_plus_numbers: false, //! allow_bare_decimal_point_numbers: false, //! allow_non_finite_numbers: false, +//! allow_extended_string_escapes: false, //! })?; //! # Ok(()) //! # } diff --git a/src/parse_to_ast.rs b/src/parse_to_ast.rs index 5da13e0..e115b7a 100644 --- a/src/parse_to_ast.rs +++ b/src/parse_to_ast.rs @@ -67,6 +67,8 @@ pub struct ParseOptions { pub allow_bare_decimal_point_numbers: bool, /// Allow the numbers Infinity, -Infinity and NaN (defaults to `true`). pub allow_non_finite_numbers: bool, + /// Allow JSON5 string escapes like \x41, \v and line continuations (defaults to `true`). + pub allow_extended_string_escapes: bool, } impl Default for ParseOptions { @@ -81,6 +83,7 @@ impl Default for ParseOptions { allow_unary_plus_numbers: true, allow_bare_decimal_point_numbers: true, allow_non_finite_numbers: true, + allow_extended_string_escapes: true, } } } @@ -261,6 +264,7 @@ pub fn parse_to_ast<'a>( allow_unary_plus_numbers: parse_options.allow_unary_plus_numbers, allow_bare_decimal_point_numbers: parse_options.allow_bare_decimal_point_numbers, allow_non_finite_numbers: parse_options.allow_non_finite_numbers, + allow_extended_string_escapes: parse_options.allow_extended_string_escapes, }, ), comments: match collect_options.comments { @@ -640,6 +644,7 @@ mod tests { allow_unary_plus_numbers: false, allow_bare_decimal_point_numbers: false, allow_non_finite_numbers: false, + allow_extended_string_escapes: false, } } diff --git a/src/parse_to_value.rs b/src/parse_to_value.rs index 5720021..391c36f 100644 --- a/src/parse_to_value.rs +++ b/src/parse_to_value.rs @@ -361,6 +361,7 @@ mod tests { allow_unary_plus_numbers: false, allow_bare_decimal_point_numbers: false, allow_non_finite_numbers: false, + allow_extended_string_escapes: false, } } diff --git a/src/parser.rs b/src/parser.rs index 67ee6e3..2acee04 100644 --- a/src/parser.rs +++ b/src/parser.rs @@ -49,6 +49,7 @@ impl<'a> JsoncParser<'a> { allow_unary_plus_numbers: options.allow_unary_plus_numbers, allow_bare_decimal_point_numbers: options.allow_bare_decimal_point_numbers, allow_non_finite_numbers: options.allow_non_finite_numbers, + allow_extended_string_escapes: options.allow_extended_string_escapes, }, ), text, diff --git a/src/scanner.rs b/src/scanner.rs index 537389f..c02371e 100644 --- a/src/scanner.rs +++ b/src/scanner.rs @@ -18,6 +18,7 @@ pub struct Scanner<'a> { allow_unary_plus_numbers: bool, allow_bare_decimal_point_numbers: bool, allow_non_finite_numbers: bool, + allow_extended_string_escapes: bool, } /// Options for the scanner. @@ -33,6 +34,8 @@ pub struct ScannerOptions { pub allow_bare_decimal_point_numbers: bool, /// Allow the numbers Infinity, -Infinity and NaN (defaults to `true`). pub allow_non_finite_numbers: bool, + /// Allow JSON5 string escapes like \x41, \v and line continuations (defaults to `true`). + pub allow_extended_string_escapes: bool, } impl Default for ScannerOptions { @@ -43,6 +46,7 @@ impl Default for ScannerOptions { allow_unary_plus_numbers: true, allow_bare_decimal_point_numbers: true, allow_non_finite_numbers: true, + allow_extended_string_escapes: true, } } } @@ -61,6 +65,7 @@ impl<'a> Scanner<'a> { allow_unary_plus_numbers: options.allow_unary_plus_numbers, allow_bare_decimal_point_numbers: options.allow_bare_decimal_point_numbers, allow_non_finite_numbers: options.allow_non_finite_numbers, + allow_extended_string_escapes: options.allow_extended_string_escapes, } } @@ -208,7 +213,8 @@ impl<'a> Scanner<'a> { } // slow path: handle escape sequences via CharProvider - crate::string::parse_string_with_char_provider(self) + let allow_extended_escapes = self.allow_extended_string_escapes; + crate::string::parse_string_with_char_provider(self, allow_extended_escapes) .map(Token::String) // todo(dsherret): don't convert the error kind to a string here .map_err(|err| self.create_error_for_start(err.byte_index, ParseErrorKind::String(err.kind))) @@ -329,13 +335,10 @@ impl<'a> Scanner<'a> { let start_byte_index = self.byte_index + 1; self.byte_index += 1; - // scan byte-by-byte for newline; \n (0x0A) and \r (0x0D) are ASCII - // and can never appear as UTF-8 continuation bytes + // scan byte-by-byte for \n or \r, which are ASCII and can never appear as UTF-8 continuation bytes; + // U+2028 and U+2029 do not end the comment, matching other JSONC parsers while let Some(&b) = self.bytes.get(self.byte_index) { - if b == b'\n' { - break; - } - if b == b'\r' && self.bytes.get(self.byte_index + 1) == Some(&b'\n') { + if b == b'\n' || b == b'\r' { break; } self.byte_index += 1; @@ -525,9 +528,10 @@ mod tests { #[test] fn it_errors_escaping_single_quote_in_double_quote() { - assert_has_error( + assert_has_error_with_options( r#""t\'est""#, "Invalid escape in double quote string on line 1 column 3", + &no_extended_string_escapes(), ); } @@ -546,12 +550,77 @@ mod tests { #[test] fn it_errors_escaping_double_quote_in_single_quote() { - assert_has_error( + assert_has_error_with_options( r#"'t\"est'"#, "Invalid escape in single quote string on line 1 column 3", + &no_extended_string_escapes(), ); } + #[test] + fn it_tokenizes_extended_string_escapes() { + assert_has_tokens( + "\"it\\'s\", 'say \\\"hi\\\"', '\\v\\0\\x41\\a', 'a\\\nb', 'a\\\rb', 'a\\\r\nb', 'a\\\u{2028}b', 'a\\\u{2029}b'", + vec![ + Token::String(Cow::Borrowed("it's")), + Token::Comma, + Token::String(Cow::Borrowed("say \"hi\"")), + Token::Comma, + Token::String(Cow::Borrowed("\u{0B}\0Aa")), + Token::Comma, + Token::String(Cow::Borrowed("ab")), + Token::Comma, + Token::String(Cow::Borrowed("ab")), + Token::Comma, + Token::String(Cow::Borrowed("ab")), + Token::Comma, + Token::String(Cow::Borrowed("ab")), + Token::Comma, + Token::String(Cow::Borrowed("ab")), + ], + ); + } + + #[test] + fn it_errors_invalid_extended_string_escapes() { + assert_has_error(r#""\1""#, "Invalid escape on line 1 column 2"); + assert_has_error(r#""\01""#, "Invalid escape on line 1 column 2"); + assert_has_error(r#""\x4""#, "Expected two hex digits on line 1 column 2"); + assert_has_error_with_options( + r#""\v""#, + "Invalid escape on line 1 column 2", + &no_extended_string_escapes(), + ); + assert_has_error_with_options( + "\"a\\\nb\"", + "Invalid escape on line 1 column 3", + &no_extended_string_escapes(), + ); + } + + #[test] + fn it_ends_comment_line_at_a_lone_carriage_return() { + assert_has_tokens( + "//a\r1,//b\u{2028}2\n3", + vec![ + Token::CommentLine("a"), + Token::Number("1"), + Token::Comma, + Token::CommentLine("b\u{2028}2"), + Token::Number("3"), + ], + ); + assert_has_error("//x\r@", "Unexpected token on line 2 column 1"); + assert_has_error("//x\r\n@", "Unexpected token on line 2 column 1"); + } + + fn no_extended_string_escapes() -> ScannerOptions { + ScannerOptions { + allow_extended_string_escapes: false, + ..Default::default() + } + } + #[test] fn it_errors_for_word_starting_with_invalid_token() { assert_has_error(r#"{ &test }"#, "Unexpected token on line 1 column 3"); @@ -745,7 +814,12 @@ mod tests { } fn assert_has_error(text: &str, message: &str) { - let mut scanner = Scanner::new(text, &Default::default()); + assert_has_error_with_options(text, message, &Default::default()); + } + + #[track_caller] + fn assert_has_error_with_options(text: &str, message: &str, options: &ScannerOptions) { + let mut scanner = Scanner::new(text, options); let mut error_message = String::new(); loop { diff --git a/src/serde.rs b/src/serde.rs index 0254b1e..2647f82 100644 --- a/src/serde.rs +++ b/src/serde.rs @@ -885,6 +885,7 @@ mod tests { allow_unary_plus_numbers: false, allow_bare_decimal_point_numbers: false, allow_non_finite_numbers: false, + allow_extended_string_escapes: false, }, ) } diff --git a/src/string.rs b/src/string.rs index ecaf2e3..4e916fd 100644 --- a/src/string.rs +++ b/src/string.rs @@ -10,6 +10,7 @@ pub enum ParseStringErrorKind { InvalidEscapeInSingleQuoteString, InvalidEscapeInDoubleQuoteString, ExpectedFourHexDigits, + ExpectedTwoHexDigits, InvalidUnicodeEscapeSequence(String), InvalidEscape, UnterminatedStringLiteral, @@ -29,6 +30,9 @@ impl std::fmt::Display for ParseStringErrorKind { ParseStringErrorKind::ExpectedFourHexDigits => { write!(f, "Expected four hex digits") } + ParseStringErrorKind::ExpectedTwoHexDigits => { + write!(f, "Expected two hex digits") + } ParseStringErrorKind::InvalidUnicodeEscapeSequence(value) => { write!( f, @@ -92,11 +96,13 @@ pub fn parse_string(text: &str) -> Result, ParseStringError> { chars, }; - parse_string_with_char_provider(&mut provider) + // the scanner already validated this text, or the caller set it raw, so decode it leniently + parse_string_with_char_provider(&mut provider, true) } pub fn parse_string_with_char_provider<'a, T: CharProvider<'a>>( chars: &mut T, + allow_extended_escapes: bool, ) -> Result, ParseStringError> { debug_assert!( chars.current_char() == Some('\'') || chars.current_char() == Some('"'), @@ -113,6 +119,17 @@ pub fn parse_string_with_char_provider<'a, T: CharProvider<'a>>( while let Some(current_char) = chars.move_next_char() { if last_was_backslash { let escape_start = chars.byte_index() - 1; // -1 for backslash + if allow_extended_escapes && let Some(decoded) = parse_extended_escape(chars, current_char, escape_start)? { + let previous_text = &chars.text()[last_start_byte_index..escape_start]; + let text = text.get_or_insert_with(String::new); + text.push_str(previous_text); + if let Some(decoded) = decoded { + text.push(decoded); + } + last_start_byte_index = chars.byte_index() + chars.current_char().map(|c| c.len_utf8()).unwrap_or(0); + last_was_backslash = false; + continue; + } match current_char { '"' | '\'' | '\\' | '/' | 'b' | 'f' | 'u' | 'r' | 'n' | 't' => { if current_char == '"' { @@ -188,6 +205,56 @@ pub fn parse_string_with_char_provider<'a, T: CharProvider<'a>>( } } +// `Ok(None)` leaves the escape to the JSON path; `Ok(Some(None))` is a line continuation, which decodes to nothing +fn parse_extended_escape<'a, T: CharProvider<'a>>( + chars: &mut T, + current_char: char, + escape_start: usize, +) -> Result>, ParseStringError> { + let invalid_escape = || ParseStringError { + byte_index: escape_start, + kind: ParseStringErrorKind::InvalidEscape, + }; + let decoded = match current_char { + 'v' => Some('\u{0B}'), + '0' => { + // `\0` must not be followed by a digit, which would make it an octal escape + let next_index = chars.byte_index() + 1; + if chars.text()[next_index..].starts_with(|c: char| c.is_ascii_digit()) { + return Err(invalid_escape()); + } + Some('\0') + } + 'x' => { + let mut value = 0; + for _ in 0..2 { + match chars.move_next_char().and_then(|c| c.to_digit(16)) { + Some(digit) => value = value * 16 + digit, + None => { + return Err(ParseStringError { + byte_index: escape_start, + kind: ParseStringErrorKind::ExpectedTwoHexDigits, + }); + } + } + } + // two hex digits are at most 0xFF, which is always a valid char + Some(char::from_u32(value).unwrap()) + } + '\n' | '\u{2028}' | '\u{2029}' => None, + '\r' => { + if chars.text()[chars.byte_index() + 1..].starts_with('\n') { + chars.move_next_char(); + } + None + } + '1'..='9' => return Err(invalid_escape()), + '\\' | '/' | 'b' | 'f' | 'u' | 'r' | 'n' | 't' => return Ok(None), + _ => Some(current_char), + }; + Ok(Some(decoded)) +} + fn read_four_hex_digits<'a, T: CharProvider<'a>>( chars: &mut T, buf: &mut [u8; 4], From 1cbd90e51c4d00b9e6eba7ec47d555acbe4aaa6f Mon Sep 17 00:00:00 2001 From: leemr Date: Fri, 25 Sep 2026 14:45:03 -0400 Subject: [PATCH 3/4] feat: support `$` and keywords in unquoted property names JSON5 property names follow ECMAScript identifiers. So `$` can appear anywhere, and the reserved words `true`, `false` and `null` are valid names: `{ $id: 1 }` and `{ true: 1 }` now parse, in both the AST and the serde parsers. `allow_loose_object_property_names` already gates unquoted names, so this adds no new switch. A name that starts with a keyword, like `true$` or `NaN$`, now scans as one word. --- src/parse_to_ast.rs | 36 +++++++++++++++++++---- src/parse_to_value.rs | 33 ++++++++++++++++++++++ src/parser.rs | 66 ++++++++++++++++--------------------------- src/scanner.rs | 7 +++-- src/serde.rs | 11 ++++++++ src/tokens.rs | 11 ++++++++ 6 files changed, 114 insertions(+), 50 deletions(-) diff --git a/src/parse_to_ast.rs b/src/parse_to_ast.rs index e115b7a..c50898c 100644 --- a/src/parse_to_ast.rs +++ b/src/parse_to_ast.rs @@ -336,11 +336,11 @@ fn parse_object<'a>(context: &mut Context<'a>) -> Result, ParseError> Some(Token::String(prop_name)) => { properties.push(parse_object_property(context, PropName::String(prop_name))?); } - Some(Token::Word(prop_name)) | Some(Token::Number(prop_name)) => { - properties.push(parse_object_property(context, PropName::Word(prop_name))?); - } None => return Err(context.create_error_for_current_range(ParseErrorKind::UnterminatedObject)), - _ => return Err(context.create_error(ParseErrorKind::UnexpectedTokenInObject)), + Some(token) => match token.as_loose_property_name() { + Some(prop_name) => properties.push(parse_object_property(context, PropName::Word(prop_name))?), + None => return Err(context.create_error(ParseErrorKind::UnexpectedTokenInObject)), + }, } // skip the comma @@ -354,7 +354,9 @@ fn parse_object<'a>(context: &mut Context<'a>) -> Result, ParseError> return Err(context.create_error_for_range(comma_range, ParseErrorKind::TrailingCommasNotAllowed)); } } - Some(Token::String(_) | Token::Word(_) | Token::Number(_)) if !context.allow_missing_commas => { + Some(Token::String(_) | Token::Word(_) | Token::Number(_) | Token::Boolean(_) | Token::Null) + if !context.allow_missing_commas => + { let range = Range { start: after_value_end, end: after_value_end, @@ -577,6 +579,11 @@ mod tests { r#"{ word: 5 }"#, "Expected string for object property on line 1 column 3", ); + assert_has_strict_error( + r#"{ true: 5 }"#, + "Expected string for object property on line 1 column 3", + ); + assert_has_strict_error(r#"{ "a": 1 null: 2 }"#, "Expected comma on line 1 column 9"); } #[test] @@ -744,6 +751,25 @@ mod tests { assert_eq!(number_value.value, "+42"); } + #[test] + fn it_should_parse_keywords_as_loose_property_names() { + let result = parse_to_ast( + r#"{ true: 1, false: null null: true }"#, + &Default::default(), + &Default::default(), + ) + .unwrap(); + let value = result.value.unwrap(); + let names = value + .as_object() + .unwrap() + .properties + .iter() + .map(|p| p.name.as_str()) + .collect::>(); + assert_eq!(names, vec!["true", "false", "null"]); + } + #[test] fn it_should_parse_json5_numbers_and_keep_non_finite_words_as_keys() { let result = parse_to_ast( diff --git a/src/parse_to_value.rs b/src/parse_to_value.rs index 391c36f..5c852e1 100644 --- a/src/parse_to_value.rs +++ b/src/parse_to_value.rs @@ -578,6 +578,39 @@ mod tests { assert_eq!(obj.get_number("CP_CanFuncReqId").unwrap(), "0x7DF"); } + #[test] + fn it_should_parse_keywords_as_loose_property_names() { + let value = parse_to_value(r#"{ true: 1, false: 2 null: 3 }"#, &Default::default()) + .unwrap() + .unwrap(); + let obj = match &value { + JsonValue::Object(o) => o, + _ => panic!("Expected object"), + }; + assert_eq!(obj.get_number("true").unwrap(), "1"); + assert_eq!(obj.get_number("false").unwrap(), "2"); + assert_eq!(obj.get_number("null").unwrap(), "3"); + } + + #[test] + fn it_should_parse_loose_property_names_with_dollar_signs() { + let value = parse_to_value( + r#"{ $: 1, _$_: 2, $_$hello123: 3, NaN$: 4, true$: 5 }"#, + &Default::default(), + ) + .unwrap() + .unwrap(); + let obj = match &value { + JsonValue::Object(o) => o, + _ => panic!("Expected object"), + }; + assert_eq!(obj.get_number("$").unwrap(), "1"); + assert_eq!(obj.get_number("_$_").unwrap(), "2"); + assert_eq!(obj.get_number("$_$hello123").unwrap(), "3"); + assert_eq!(obj.get_number("NaN$").unwrap(), "4"); + assert_eq!(obj.get_number("true$").unwrap(), "5"); + } + #[test] fn it_should_parse_loose_property_names_starting_with_infinity_or_nan() { let value = parse_to_value(r#"{ Infinity_count: 1, NaN-key: 2 }"#, &Default::default()) diff --git a/src/parser.rs b/src/parser.rs index 2acee04..429b33a 100644 --- a/src/parser.rs +++ b/src/parser.rs @@ -129,7 +129,8 @@ impl<'a> JsoncParser<'a> { /// between entries. Pass `first = true` for the first entry. pub fn scan_object_entry(&mut self, first: bool) -> Result>, ParseError> { if first { - return self.scan_object_key(); + let token = self.scan()?; + return self.object_key(token); } let after_value_end = self.scanner.token_end(); @@ -137,7 +138,8 @@ impl<'a> JsoncParser<'a> { match token { Some(Token::Comma) => { let comma_range = Range::new(self.scanner.token_start(), self.scanner.token_end()); - let key = self.scan_object_key()?; + let token = self.scan()?; + let key = self.object_key(token)?; if key.is_none() && !self.allow_trailing_commas { return Err( self @@ -147,19 +149,10 @@ impl<'a> JsoncParser<'a> { } Ok(key) } - Some(Token::CloseBrace) => Ok(None), - Some(Token::String(s)) if self.allow_missing_commas => Ok(Some(ObjectKey::String(s))), - Some(Token::Word(s) | Token::Number(s)) if self.allow_missing_commas => { - if !self.allow_loose_object_property_names { - return Err( - self - .scanner - .create_error_for_current_token(ParseErrorKind::ExpectedStringObjectProperty), - ); - } - Ok(Some(ObjectKey::Word(s))) - } - Some(Token::String(_) | Token::Word(_) | Token::Number(_)) => { + Some(ref token) + if !self.allow_missing_commas + && (matches!(token, Token::String(_)) || token.as_loose_property_name().is_some()) => + { let range = Range::new(after_value_end, after_value_end); Err( self @@ -167,16 +160,7 @@ impl<'a> JsoncParser<'a> { .create_error_for_range(range, ParseErrorKind::ExpectedComma), ) } - None => Err( - self - .scanner - .create_error_for_current_token(ParseErrorKind::UnterminatedObject), - ), - _ => Err( - self - .scanner - .create_error_for_current_token(ParseErrorKind::UnexpectedTokenInObject), - ), + token => self.object_key(token), } } @@ -222,30 +206,28 @@ impl<'a> JsoncParser<'a> { } } - fn scan_object_key(&mut self) -> Result>, ParseError> { - match self.scan()? { + fn object_key(&self, token: Option>) -> Result>, ParseError> { + match token { Some(Token::CloseBrace) => Ok(None), Some(Token::String(s)) => Ok(Some(ObjectKey::String(s))), - Some(Token::Word(s) | Token::Number(s)) => { - if !self.allow_loose_object_property_names { - return Err( - self - .scanner - .create_error_for_current_token(ParseErrorKind::ExpectedStringObjectProperty), - ); - } - Ok(Some(ObjectKey::Word(s))) - } None => Err( self .scanner .create_error_for_current_token(ParseErrorKind::UnterminatedObject), ), - _ => Err( - self - .scanner - .create_error_for_current_token(ParseErrorKind::UnexpectedTokenInObject), - ), + Some(token) => match token.as_loose_property_name() { + Some(_) if !self.allow_loose_object_property_names => Err( + self + .scanner + .create_error_for_current_token(ParseErrorKind::ExpectedStringObjectProperty), + ), + Some(s) => Ok(Some(ObjectKey::Word(s))), + None => Err( + self + .scanner + .create_error_for_current_token(ParseErrorKind::UnexpectedTokenInObject), + ), + }, } } } diff --git a/src/scanner.rs b/src/scanner.rs index c02371e..5f213a1 100644 --- a/src/scanner.rs +++ b/src/scanner.rs @@ -401,9 +401,9 @@ impl<'a> Scanner<'a> { if &self.bytes[self.byte_index..end] != text_bytes { return false; } - // ensure the word is not followed by an alphanumeric character + // ensure the word is not followed by an alphanumeric character or `$` if let Some(&next_byte) = self.bytes.get(end) { - if next_byte.is_ascii_alphanumeric() { + if next_byte.is_ascii_alphanumeric() || next_byte == b'$' { return false; } // check non-ASCII alphanumeric @@ -437,7 +437,8 @@ impl<'a> Scanner<'a> { if b.is_ascii_whitespace() || b == b':' { break; } - if b.is_ascii_alphanumeric() || b == b'-' || b == b'_' { + // `$` is valid anywhere in a JSON5 (ECMAScript) identifier + if b.is_ascii_alphanumeric() || b == b'-' || b == b'_' || b == b'$' { self.byte_index += 1; } else { return Err(self.create_error_for_current_token(ParseErrorKind::UnexpectedToken)); diff --git a/src/serde.rs b/src/serde.rs index 2647f82..efed312 100644 --- a/src/serde.rs +++ b/src/serde.rs @@ -523,6 +523,17 @@ mod tests { assert_eq!(result, SerdeValue::Object(expected_value)); } + #[test] + fn it_should_parse_keywords_as_loose_property_names() { + let result = parse_to_serde_value::(r#"{ true: 1, false: 2 null: 3 }"#, &Default::default()).unwrap(); + assert_eq!(result, serde_json::json!({ "true": 1, "false": 2, "null": 3 })); + assert_has_strict_error( + r#"{ null: 1 }"#, + "Expected string for object property on line 1 column 3", + ); + assert_has_strict_error(r#"{ "a": 1 true: 2 }"#, "Expected comma on line 1 column 9"); + } + #[test] fn it_should_parse_unary_plus_numbers() { let result = parse_to_serde_value::( diff --git a/src/tokens.rs b/src/tokens.rs index 8096bb0..5b11616 100644 --- a/src/tokens.rs +++ b/src/tokens.rs @@ -45,6 +45,17 @@ impl<'a> Token<'a> { } } + /// The text of a token that can be an unquoted property name, which in JSON5 includes `true`, `false` and `null`. + pub(crate) fn as_loose_property_name(&self) -> Option<&'a str> { + match self { + Token::Word(value) | Token::Number(value) => Some(value), + Token::Boolean(true) => Some("true"), + Token::Boolean(false) => Some("false"), + Token::Null => Some("null"), + _ => None, + } + } + /// Whether this token can begin a JSON value. pub(crate) fn is_value_start(&self) -> bool { matches!( From 612749c84a964f16dbb897b5d4d870efa7047974 Mon Sep 17 00:00:00 2001 From: David Sherret Date: Sun, 27 Sep 2026 12:51:39 -0400 Subject: [PATCH 4/4] fix(cst): treat a lone carriage return as a newline A lone `\r` now ends a line comment, but the CST stored it as whitespace, so edits like `append` and sorting treated the comment as still open and could place a new value inside the comment. --- src/cst/mod.rs | 39 +++++++++++++++++++++++++++++++++++++-- 1 file changed, 37 insertions(+), 2 deletions(-) diff --git a/src/cst/mod.rs b/src/cst/mod.rs index 7d3871c..cddc86d 100644 --- a/src/cst/mod.rs +++ b/src/cst/mod.rs @@ -2372,9 +2372,10 @@ pub enum CstNewlineKind { #[default] LineFeed, CarriageReturnLineFeed, + CarriageReturn, } -/// Newline character (Lf or crlf). +/// Newline character (Lf, crlf, or cr). #[derive(Debug, Clone)] pub struct CstNewline(Rc>>); @@ -2385,7 +2386,7 @@ impl CstNewline { Self(CstValueInner::new(kind)) } - /// Whether this is a line feed (LF) or carriage return line feed (CRLF). + /// Whether this is a line feed (LF), carriage return line feed (CRLF), or carriage return (CR). pub fn kind(&self) -> CstNewlineKind { self.0.borrow().value } @@ -2407,6 +2408,7 @@ impl Display for CstNewline { #[allow(clippy::write_with_newline)] // better to be explicit CstNewlineKind::LineFeed => write!(f, "\n"), CstNewlineKind::CarriageReturnLineFeed => write!(f, "\r\n"), + CstNewlineKind::CarriageReturn => write!(f, "\r"), } } } @@ -2614,6 +2616,10 @@ impl<'a> CstBuilder<'a> { container.raw_append_child(CstNewline::new(CstNewlineKind::CarriageReturnLineFeed).into()); last_found_index = i + 2; chars.next(); // move past the \n + } else if c == '\r' { + maybe_add_previous_text(last_found_index, i); + container.raw_append_child(CstNewline::new(CstNewlineKind::CarriageReturn).into()); + last_found_index = i + 1; } else if c == '\n' { maybe_add_previous_text(last_found_index, i); container.raw_append_child(CstNewline::new(CstNewlineKind::LineFeed).into()); @@ -3874,6 +3880,7 @@ mod test { use pretty_assertions::assert_eq; use crate::cst::CstInputValue; + use crate::cst::CstNewlineKind; use crate::cst::TrailingCommaMode; use crate::json; @@ -5533,4 +5540,32 @@ value3: true .unwrap(); assert_eq!(decoded, "key\\with\\backslash"); } + + #[test] + fn lone_carriage_return_ends_line_comment() { + // a lone \r ends a line comment, so edits should behave the same as with \n + #[track_caller] + fn run_test(json: &str, edit: impl Fn(&CstRootNode)) { + let lf_cst = build_cst(json); + edit(&lf_cst); + let cr_cst = build_cst(&json.replace('\n', "\r")); + assert_eq!(cr_cst.newline_kind(), CstNewlineKind::CarriageReturn); + edit(&cr_cst); + assert_eq!(cr_cst.to_string(), lf_cst.to_string().replace('\n', "\r")); + } + + run_test("[ // note\n]", |cst| { + cst.array_value().unwrap().append(json!(3)); + }); + run_test("[\n 1 // one\n]", |cst| { + cst.array_value().unwrap().append(json!(3)); + }); + run_test("{\n \"b\": 1, // bee\n \"a\": 2\n}", |cst| { + cst + .object_value() + .unwrap() + .sort_properties() + .by_key(|prop| prop.decoded_name()); + }); + } }