From 8ae6169f567a9adcbcc6fdc587bebb24bc6a698e Mon Sep 17 00:00:00 2001 From: Stefan Date: Wed, 7 Feb 2024 01:01:04 +0100 Subject: [PATCH] feat: allow strings to be escaped; --- core/expression/src/lexer/lexer.rs | 5 +- core/expression/src/parser/mod.rs | 1 + core/expression/src/parser/parser.rs | 7 +- .../expression/src/parser/sanitised_string.rs | 120 ++++++++++++++++++ 4 files changed, 131 insertions(+), 2 deletions(-) create mode 100644 core/expression/src/parser/sanitised_string.rs diff --git a/core/expression/src/lexer/lexer.rs b/core/expression/src/lexer/lexer.rs index eabf8fd4..98a7ee16 100644 --- a/core/expression/src/lexer/lexer.rs +++ b/core/expression/src/lexer/lexer.rs @@ -83,12 +83,15 @@ impl<'arena, 'self_ref> Scanner<'arena, 'self_ref> { let (start, opener) = self.next()?; let end: usize; + let mut is_escaped = false; loop { let (e, c) = self.next()?; - if c == opener { + if c == opener && !is_escaped { end = e; break; } + + is_escaped = c == '\\' && !is_escaped; } self.push(Token { diff --git a/core/expression/src/parser/mod.rs b/core/expression/src/parser/mod.rs index 1d5b3afe..10db7be9 100644 --- a/core/expression/src/parser/mod.rs +++ b/core/expression/src/parser/mod.rs @@ -10,6 +10,7 @@ mod builtin; mod constants; mod error; mod parser; +mod sanitised_string; mod standard; mod unary; diff --git a/core/expression/src/parser/parser.rs b/core/expression/src/parser/parser.rs index 3306284f..0b0b1f8b 100644 --- a/core/expression/src/parser/parser.rs +++ b/core/expression/src/parser/parser.rs @@ -10,6 +10,7 @@ use crate::lexer::{Bracket, ComparisonOperator, Identifier, Operator, Token, Tok use crate::parser::ast::Node; use crate::parser::builtin::{Arity, BuiltInFunction}; use crate::parser::error::{ParserError, ParserResult}; +use crate::parser::sanitised_string::SanitisedString; use crate::parser::standard::Standard; use crate::parser::unary::Unary; @@ -148,7 +149,11 @@ impl<'arena, 'token_ref, Flavor> Parser<'arena, 'token_ref, Flavor> { } self.next()?; - Ok(Some(self.node(Node::String(current_token.value)))) + + let sanitised = SanitisedString::from(current_token.value); + Ok(Some( + self.node(Node::String(sanitised.into_bump_str(self.bump))), + )) } pub(crate) fn bool(&self) -> ParserResult>> { diff --git a/core/expression/src/parser/sanitised_string.rs b/core/expression/src/parser/sanitised_string.rs new file mode 100644 index 00000000..e9ab667a --- /dev/null +++ b/core/expression/src/parser/sanitised_string.rs @@ -0,0 +1,120 @@ +use bumpalo::collections::String as BumpString; +use bumpalo::Bump; + +#[derive(Debug)] +pub(crate) struct SanitisedString<'str>(&'str str); + +impl<'str> From<&'str str> for SanitisedString<'str> { + fn from(value: &'str str) -> Self { + Self(value) + } +} + +impl<'str> SanitisedString<'str> { + fn contains_escapes(&self) -> bool { + self.0.contains('\\') + } + + pub(crate) fn into_bump_str<'arena>(self, bump: &'arena Bump) -> &'arena str + where + 'str: 'arena, + { + if !self.contains_escapes() { + return self.0; + } + + let mut result = BumpString::new_in(bump); + let mut chars = self.0.chars().peekable(); + while let Some(c) = chars.next() { + if c != '\\' { + result.push(c); + continue; + } + + if let Some(&'\\') = chars.peek() { + result.push('\\'); + chars.next(); + } + } + + return result.into_bump_str(); + } +} + +#[cfg(test)] +mod tests { + use crate::parser::sanitised_string::SanitisedString; + use bumpalo::Bump; + + fn sanitise_string(data: &str) -> String { + let bump = Bump::new(); + let string = SanitisedString::from(data).into_bump_str(&bump); + + String::from(string) + } + + fn test_case(data: &str, expected: &str) { + let sanitised = sanitise_string(data); + assert_eq!(sanitised.as_str(), expected); + } + + #[test] + fn it_handles_varied_normal_strings() { + test_case("1234567890", "1234567890"); + test_case("Abc123Xyz890", "Abc123Xyz890"); + test_case("!@# $%^ &*()", "!@# $%^ &*()"); + test_case("", ""); + } + + #[test] + fn it_handles_various_strings() { + // Test case with single backslashes that should be removed + test_case("This is a test\\ string.", "This is a test string."); + + // Test case with double backslashes that should be reduced to single backslashes + test_case( + "A string with \\\\ double backslashes.", + "A string with \\ double backslashes.", + ); + + // Test case with a mixture of single and double backslashes + test_case( + "Mix of \\single and \\\\double backslashes.", + "Mix of single and \\double backslashes.", + ); + + // Test case with backslashes at the beginning and end of the string + test_case("\\Start and end\\", "Start and end"); + + // Test case with consecutive double backslashes + test_case( + "Consecutive \\\\\\\\ backslashes.", + "Consecutive \\\\ backslashes.", + ); + } + + #[test] + fn it_removes_single_backslashes() { + test_case(r#"\Start of string"#, r#"Start of string"#); + + // Single backslash at the end of the string + test_case(r#"End of string\"#, r#"End of string"#); + + // Single backslashes surrounding a word + test_case(r#"Word \with\ backslashes"#, r#"Word with backslashes"#); + + // Single backslash before a space and a character + test_case( + r#"Backslash before \ space and \c"#, + r#"Backslash before space and c"#, + ); + + // Single backslash adjacent to punctuation + test_case(r#"Punctuation\! and \?"#, r#"Punctuation! and ?"#); + + test_case( + r#"Double quotes \" and single quotes \'"#, + r#"Double quotes " and single quotes '"#, + ); + } +}