diff --git a/CMakeLists.txt b/CMakeLists.txt index 732d05a..08e6a10 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -263,6 +263,7 @@ if (BUILD_TESTS) tests/conversion.cpp tests/value.cpp tests/strict.cpp + tests/string.cpp tests/locale.cpp ) target_compile_definitions(${TEST_TARGET} PRIVATE CORPUS_DIR="${CMAKE_CURRENT_SOURCE_DIR}/tests/corpus") diff --git a/include/numen/numen.hpp b/include/numen/numen.hpp index 2807021..3d09128 100644 --- a/include/numen/numen.hpp +++ b/include/numen/numen.hpp @@ -211,7 +211,8 @@ template struct ConversionOf { }; struct Conversion { - std::variant, ConversionOf> sides; + // the string alternative names the converter that ran, e.g. "upper" + std::variant, ConversionOf, ConversionOf> sides; // locale currency conversion bool implicit = false; diff --git a/src/numen/fn.cpp b/src/numen/fn.cpp index 0df5964..d2ac33c 100644 --- a/src/numen/fn.cpp +++ b/src/numen/fn.cpp @@ -1,7 +1,9 @@ #include "fn.hpp" #include "computed.hpp" #include "numen/numen.hpp" +#include "utils.hpp" #include +#include #include #include #include @@ -334,6 +336,26 @@ FunctionDatabase makeBuiltin() { }); } + db.addConverter("upper", [](const FunctionCtx &ctx) { + ctx.expectArgs(1); + if (auto str = ctx.args.front().asStr()) { + std::string out = *str; + upperCase(out); + return Computed{out}; + } + throw std::runtime_error("Invalid type"); + }); + + db.addConverter("lower", [](const FunctionCtx &ctx) { + ctx.expectArgs(1); + if (auto str = ctx.args.front().asStr()) { + std::string out = *str; + lowerCase(out); + return Computed{out}; + } + throw std::runtime_error("Invalid type"); + }); + db.addConverter("json", [](const FunctionCtx &ctx) { ctx.expectArgs(1); auto &arg = ctx.args[0]; @@ -344,11 +366,6 @@ FunctionDatabase makeBuiltin() { if constexpr (std::is_same_v) { return Computed{value.toRFC3339()}; - } - - // FIXME: technically {:?} is not the same as JSON escaping (I'm pretty sure) - else if constexpr (std::is_same_v) { - return Computed{std::format("{:?}", value)}; } else if constexpr (std::is_same_v) { return Computed{value}; } else { diff --git a/src/numen/interpreter.cpp b/src/numen/interpreter.cpp index 009b9f7..b89d093 100644 --- a/src/numen/interpreter.cpp +++ b/src/numen/interpreter.cpp @@ -201,6 +201,8 @@ class Interpreter { throw std::runtime_error(std::format("Cannot convert a {} to that", v.valueTypeName())); } else if constexpr (std::is_same_v) { return Computed{.value = Num{value}}; + } else if constexpr (std::is_same_v) { + return Computed{.value = std::string{value.data}}; } else if constexpr (std::is_same_v) { auto lhs = computeExpr(*value.lhs); if (auto n = lhs.asNumber(); n && value.op == "k") { n->n = n->n.toDouble() * 1e3; } @@ -810,6 +812,11 @@ class Interpreter { auto r = (*handler)(FunctionCtx{.name = fn.name, .args = computedArgs}); r.explicitlyConverted = std::ranges::any_of(computedArgs, &Computed::explicitlyConverted); + + if (std::ranges::contains(FunctionDatabase::builtin().converterNames(), fn.name)) { + r.conversion = Conversion{.sides = ConversionOf{.to = std::string{fn.name}}}; + } + return r; } diff --git a/src/numen/lexer.cpp b/src/numen/lexer.cpp index b84f4e8..29b9898 100644 --- a/src/numen/lexer.cpp +++ b/src/numen/lexer.cpp @@ -30,6 +30,8 @@ std::optional Lexer::next() { unsigned nfrac = 0; int expSign = 1; unsigned expValue = 0; + std::string stringLit{}; + char stringLiteralOpener = 0; const auto getSelection = [&]() -> std::string_view { return m_data.substr(startPos, m_cursor - startPos); @@ -43,7 +45,8 @@ std::optional Lexer::next() { }; const auto makeToken = [&](TokenType type, TokenData data) { - return Token{.raw = getSelection(), .type = type, .data = data, .start = startPos, .end = m_cursor}; + return Token{ + .raw = getSelection(), .type = type, .data = std::move(data), .start = startPos, .end = m_cursor}; }; constexpr auto isValidChar = [](char c) { @@ -71,6 +74,8 @@ std::optional Lexer::next() { } case State::String: return makeToken(TokenType::String, String{.data = getSelection()}); + case State::StringLiteral: + return makeToken(TokenType::StringLiteral, StringLiteral{.data = stringLit}); case State::Operator: return makeToken(TokenType::Operator, Operator{getSelection()}); default: @@ -95,6 +100,12 @@ std::optional Lexer::next() { startPos += 1; break; } + if (c == '"' || c == '\'') { + state = State::StringLiteral; + stringLiteralOpener = c; + startPos += 1; + break; + } if (isValidChar(c)) { state = State::String; continue; @@ -197,6 +208,24 @@ std::optional Lexer::next() { if (isDigit(c) && isCalled(m_cursor)) break; return tryCommit(); } + case State::StringLiteral: { + if (c == stringLiteralOpener) { + auto tok = tryCommit(); + ++m_cursor; + return tok; + } + + if (c == '\\') { + ++m_cursor; + // for now escaping just appends the next character as is, disregarding quoting rules + // We don't parse special escape sequences. Maybe we can do it in the future if it proves useful. + if (m_cursor < m_data.size()) stringLit += m_data[m_cursor]; + } else { + stringLit += c; + } + + break; + } } ++m_cursor; diff --git a/src/numen/lexer.hpp b/src/numen/lexer.hpp index d243bfa..55bcc68 100644 --- a/src/numen/lexer.hpp +++ b/src/numen/lexer.hpp @@ -16,18 +16,30 @@ class Lexer { unsigned fromBase = 10; }; - enum class OperatorType { Add, Subtract, Multiply, Divide, Pow }; - enum class State { Reset, Number, Operator, NumberBase, NumberExponentSign, NumberExponent, String }; - enum class TokenType { String, Number, Operator }; + enum class State { + Reset, + Number, + Operator, + NumberBase, + NumberExponentSign, + NumberExponent, + String, + StringLiteral + }; + enum class TokenType { String, StringLiteral, Number, Operator }; struct String { std::string_view data; }; + struct StringLiteral { + // not a view, because of escaping + std::string data; + }; struct Operator { std::string_view op; }; - using TokenData = std::variant; + using TokenData = std::variant; struct Token { std::string_view raw; @@ -36,6 +48,9 @@ class Lexer { std::string_view::size_type start = 0; std::string_view::size_type end = 0; + template T *as() { return std::get_if(&data); } + template const T *as() const { return std::get_if(&data); } + bool isAdjacent(const Token &rhs) const { return end == rhs.start; } const Number *asNumber() const { return std::get_if(&data); } diff --git a/src/numen/parser.cpp b/src/numen/parser.cpp index b99a363..593c502 100644 --- a/src/numen/parser.cpp +++ b/src/numen/parser.cpp @@ -773,11 +773,15 @@ std::unique_ptr Parser::parseTerm() { throw std::runtime_error("Expected EOF, looks like there is nothing we can parse!"); } + if (auto str = m_lexer.peakAs()) { + m_lexer.next(); + return std::make_unique(StringLiteral{std::move(str->data)}); + } + auto expr = std::unique_ptr(); auto frontUnit = parseUnit(); if (auto tok = m_lexer.peak()) { - if (auto constant = parseConstant(tok->raw)) { m_lexer.next(); diff --git a/src/numen/parser.hpp b/src/numen/parser.hpp index 9f74f46..58eab65 100644 --- a/src/numen/parser.hpp +++ b/src/numen/parser.hpp @@ -130,9 +130,13 @@ struct PercentExpression { std::unique_ptr expr; }; +struct StringLiteral { + std::string data; +}; + struct Expression { std::variant + ConversionExpression, StringLiteral, Duration, FunctionCall, PercentExpression> data; const BinaryExpression *asBinaryExpression() const { return as(); } diff --git a/src/numen/utils.hpp b/src/numen/utils.hpp index b185b53..2d5a69b 100644 --- a/src/numen/utils.hpp +++ b/src/numen/utils.hpp @@ -14,3 +14,7 @@ inline bool equalsIgnoreCase(const auto &a, const auto &b) { inline void lowerCase(std::string &s) { std::ranges::transform(s, s.begin(), [](unsigned char c) { return static_cast(std::tolower(c)); }); } + +inline void upperCase(std::string &s) { + std::ranges::transform(s, s.begin(), [](unsigned char c) { return static_cast(std::toupper(c)); }); +} diff --git a/tests/corpus/strings.corpus b/tests/corpus/strings.corpus new file mode 100644 index 0000000..99f90d0 --- /dev/null +++ b/tests/corpus/strings.corpus @@ -0,0 +1,42 @@ +# string literals: both quote styles delimit text, backslash escapes the next character + +"hello world" => hello world +'hello world' => hello world +"mixed 'quotes'" => mixed 'quotes' +'mixed "quotes"' => mixed "quotes" + +# escaping strips the backslash and keeps the next character as is +"say \"hi\"" => say "hi" +'it\'s' => it's +"back\\slash" => back\slash +"\q\w" => qw + +# no special escape sequences +"line\nline" => linenline +"tab\there" => tabthere + +# quotes shield text that would otherwise mean something +"km" => km +'pi' => pi +"e" => e +"123" => 123 +"1 + 1" => 1 + 1 +'now' => now +"12:30" => 12:30 + +# converters work on strings, in both forms +'hello' to upper => HELLO +upper('hello') => HELLO +"HeLLo" to lower => hello +lower("HeLLo") => hello + +# strings do not mix with numbers +"abc" + 1 => !error Cannot add Number to Text +1 + "abc" => !error Cannot add Text to Number +2 * "abc" => !error Invalid operands: Number and Text +"abc" + "def" => !error Cannot add Text to Text + +# a literal still being typed is committed as is +"unfinished => unfinished +'unfinished => unfinished +"trailing\ => trailing diff --git a/tests/string.cpp b/tests/string.cpp new file mode 100644 index 0000000..4a9fc68 --- /dev/null +++ b/tests/string.cpp @@ -0,0 +1,47 @@ +#include "helpers.hpp" +#include "numen/numen.hpp" +#include + +TEST_CASE("empty string literals evaluate to empty text") { + test::assertExpr(R"("")", ""); + test::assertExpr("''", ""); +} + +TEST_CASE("string literals preserve whitespace") { + test::assertExpr(R"(" a b ")", " a b "); + test::assertExpr("'\ttab'", "\ttab"); +} + +TEST_CASE("String literal alone should be marked as non converted") { + numen::Numen calc{}; + auto res = calc.compute("'this is a string literal'"); + REQUIRE(res); + REQUIRE(res->asStr()); + REQUIRE(!res->conversion); +} + +TEST_CASE("Transformed string should be marked as converted") { + numen::Numen calc{}; + auto res = calc.compute("'this is a string literal' to upper"); + REQUIRE(res); + REQUIRE(res->asStr()); + REQUIRE(res->conversion); +} + +TEST_CASE("Converter call form marks the conversion and names the converter") { + numen::Numen calc{}; + auto res = calc.compute("upper('hello')"); + REQUIRE(res); + REQUIRE(res->asStr()); + CHECK(*res->asStr() == "HELLO"); + REQUIRE(res->conversion); + REQUIRE(res->conversion->as()); + CHECK(res->conversion->as()->to == "upper"); +} + +TEST_CASE("Regular functions do not mark a conversion") { + numen::Numen calc{}; + auto res = calc.compute("sqrt(4)"); + REQUIRE(res); + REQUIRE(!res->conversion); +}