Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -263,6 +263,7 @@ if (BUILD_TESTS)
tests/conversion.cpp
tests/value.cpp
tests/strict.cpp
tests/string.cpp
tests/locale.cpp
)
target_compile_definitions(${TEST_TARGET} PRIVATE CORPUS_DIR="${CMAKE_CURRENT_SOURCE_DIR}/tests/corpus")
Expand Down
3 changes: 2 additions & 1 deletion include/numen/numen.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -211,7 +211,8 @@ template <class T> struct ConversionOf {
};

struct Conversion {
std::variant<ConversionOf<Number::Unit>, ConversionOf<Timezone>> sides;
// the string alternative names the converter that ran, e.g. "upper"
std::variant<ConversionOf<Number::Unit>, ConversionOf<Timezone>, ConversionOf<std::string>> sides;

// locale currency conversion
bool implicit = false;
Expand Down
27 changes: 22 additions & 5 deletions src/numen/fn.cpp
Original file line number Diff line number Diff line change
@@ -1,7 +1,9 @@
#include "fn.hpp"
#include "computed.hpp"
#include "numen/numen.hpp"
#include "utils.hpp"
#include <algorithm>
#include <cctype>
#include <cmath>
#include <cstdint>
#include <format>
Expand Down Expand Up @@ -334,6 +336,26 @@ FunctionDatabase makeBuiltin() {
});
}

db.addConverter("upper", [](const FunctionCtx &ctx) {
ctx.expectArgs(1);
if (auto str = ctx.args.front().asStr()) {
std::string out = *str;
upperCase(out);
return Computed{out};
}
throw std::runtime_error("Invalid type");
});

db.addConverter("lower", [](const FunctionCtx &ctx) {
ctx.expectArgs(1);
if (auto str = ctx.args.front().asStr()) {
std::string out = *str;
lowerCase(out);
return Computed{out};
}
throw std::runtime_error("Invalid type");
});

db.addConverter("json", [](const FunctionCtx &ctx) {
ctx.expectArgs(1);
auto &arg = ctx.args[0];
Expand All @@ -344,11 +366,6 @@ FunctionDatabase makeBuiltin() {

if constexpr (std::is_same_v<T, DateTime>) {
return Computed{value.toRFC3339()};
}

// FIXME: technically {:?} is not the same as JSON escaping (I'm pretty sure)
else if constexpr (std::is_same_v<T, std::string>) {
return Computed{std::format("{:?}", value)};
} else if constexpr (std::is_same_v<T, Num>) {
return Computed{value};
} else {
Expand Down
7 changes: 7 additions & 0 deletions src/numen/interpreter.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -201,6 +201,8 @@ class Interpreter {
throw std::runtime_error(std::format("Cannot convert a {} to that", v.valueTypeName()));
} else if constexpr (std::is_same_v<T, NumberString>) {
return Computed{.value = Num{value}};
} else if constexpr (std::is_same_v<T, StringLiteral>) {
return Computed{.value = std::string{value.data}};
} else if constexpr (std::is_same_v<T, PostfixExpression>) {
auto lhs = computeExpr(*value.lhs);
if (auto n = lhs.asNumber(); n && value.op == "k") { n->n = n->n.toDouble() * 1e3; }
Expand Down Expand Up @@ -810,6 +812,11 @@ class Interpreter {

auto r = (*handler)(FunctionCtx{.name = fn.name, .args = computedArgs});
r.explicitlyConverted = std::ranges::any_of(computedArgs, &Computed::explicitlyConverted);

if (std::ranges::contains(FunctionDatabase::builtin().converterNames(), fn.name)) {
r.conversion = Conversion{.sides = ConversionOf<std::string>{.to = std::string{fn.name}}};
}

return r;
}

Expand Down
31 changes: 30 additions & 1 deletion src/numen/lexer.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,8 @@ std::optional<Lexer::Token> Lexer::next() {
unsigned nfrac = 0;
int expSign = 1;
unsigned expValue = 0;
std::string stringLit{};
char stringLiteralOpener = 0;

const auto getSelection = [&]() -> std::string_view {
return m_data.substr(startPos, m_cursor - startPos);
Expand All @@ -43,7 +45,8 @@ std::optional<Lexer::Token> Lexer::next() {
};

const auto makeToken = [&](TokenType type, TokenData data) {
return Token{.raw = getSelection(), .type = type, .data = data, .start = startPos, .end = m_cursor};
return Token{
.raw = getSelection(), .type = type, .data = std::move(data), .start = startPos, .end = m_cursor};
};

constexpr auto isValidChar = [](char c) {
Expand Down Expand Up @@ -71,6 +74,8 @@ std::optional<Lexer::Token> Lexer::next() {
}
case State::String:
return makeToken(TokenType::String, String{.data = getSelection()});
case State::StringLiteral:
return makeToken(TokenType::StringLiteral, StringLiteral{.data = stringLit});
case State::Operator:
return makeToken(TokenType::Operator, Operator{getSelection()});
default:
Expand All @@ -95,6 +100,12 @@ std::optional<Lexer::Token> Lexer::next() {
startPos += 1;
break;
}
if (c == '"' || c == '\'') {
state = State::StringLiteral;
stringLiteralOpener = c;
startPos += 1;
break;
}
if (isValidChar(c)) {
state = State::String;
continue;
Expand Down Expand Up @@ -197,6 +208,24 @@ std::optional<Lexer::Token> Lexer::next() {
if (isDigit(c) && isCalled(m_cursor)) break;
return tryCommit();
}
case State::StringLiteral: {
if (c == stringLiteralOpener) {
auto tok = tryCommit();
++m_cursor;
return tok;
}

if (c == '\\') {
++m_cursor;
// for now escaping just appends the next character as is, disregarding quoting rules
// We don't parse special escape sequences. Maybe we can do it in the future if it proves useful.
if (m_cursor < m_data.size()) stringLit += m_data[m_cursor];
} else {
stringLit += c;
}

break;
}
}

++m_cursor;
Expand Down
23 changes: 19 additions & 4 deletions src/numen/lexer.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -16,18 +16,30 @@ class Lexer {
unsigned fromBase = 10;
};

enum class OperatorType { Add, Subtract, Multiply, Divide, Pow };
enum class State { Reset, Number, Operator, NumberBase, NumberExponentSign, NumberExponent, String };
enum class TokenType { String, Number, Operator };
enum class State {
Reset,
Number,
Operator,
NumberBase,
NumberExponentSign,
NumberExponent,
String,
StringLiteral
};
enum class TokenType { String, StringLiteral, Number, Operator };

struct String {
std::string_view data;
};
struct StringLiteral {
// not a view, because of escaping
std::string data;
};
struct Operator {
std::string_view op;
};

using TokenData = std::variant<Number, String, Operator>;
using TokenData = std::variant<Number, String, StringLiteral, Operator>;

struct Token {
std::string_view raw;
Expand All @@ -36,6 +48,9 @@ class Lexer {
std::string_view::size_type start = 0;
std::string_view::size_type end = 0;

template <typename T> T *as() { return std::get_if<T>(&data); }
template <typename T> const T *as() const { return std::get_if<T>(&data); }

bool isAdjacent(const Token &rhs) const { return end == rhs.start; }

const Number *asNumber() const { return std::get_if<Number>(&data); }
Expand Down
6 changes: 5 additions & 1 deletion src/numen/parser.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -773,11 +773,15 @@ std::unique_ptr<Expression> Parser::parseTerm() {
throw std::runtime_error("Expected EOF, looks like there is nothing we can parse!");
}

if (auto str = m_lexer.peakAs<Lexer::StringLiteral>()) {
m_lexer.next();
return std::make_unique<Expression>(StringLiteral{std::move(str->data)});
}

auto expr = std::unique_ptr<Expression>();
auto frontUnit = parseUnit();

if (auto tok = m_lexer.peak()) {

if (auto constant = parseConstant(tok->raw)) {
m_lexer.next();

Expand Down
6 changes: 5 additions & 1 deletion src/numen/parser.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -130,9 +130,13 @@ struct PercentExpression {
std::unique_ptr<Expression> expr;
};

struct StringLiteral {
std::string data;
};

struct Expression {
std::variant<BinaryExpression, UnaryExpression, PostfixExpression, NumberString, DateString, UnitExpression,
ConversionExpression, Duration, FunctionCall, PercentExpression>
ConversionExpression, StringLiteral, Duration, FunctionCall, PercentExpression>
data;

const BinaryExpression *asBinaryExpression() const { return as<BinaryExpression>(); }
Expand Down
4 changes: 4 additions & 0 deletions src/numen/utils.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -14,3 +14,7 @@ inline bool equalsIgnoreCase(const auto &a, const auto &b) {
inline void lowerCase(std::string &s) {
std::ranges::transform(s, s.begin(), [](unsigned char c) { return static_cast<char>(std::tolower(c)); });
}

inline void upperCase(std::string &s) {
std::ranges::transform(s, s.begin(), [](unsigned char c) { return static_cast<char>(std::toupper(c)); });
}
42 changes: 42 additions & 0 deletions tests/corpus/strings.corpus
Original file line number Diff line number Diff line change
@@ -0,0 +1,42 @@
# string literals: both quote styles delimit text, backslash escapes the next character

"hello world" => hello world
'hello world' => hello world
"mixed 'quotes'" => mixed 'quotes'
'mixed "quotes"' => mixed "quotes"

# escaping strips the backslash and keeps the next character as is
"say \"hi\"" => say "hi"
'it\'s' => it's
"back\\slash" => back\slash
"\q\w" => qw

# no special escape sequences
"line\nline" => linenline
"tab\there" => tabthere

# quotes shield text that would otherwise mean something
"km" => km
'pi' => pi
"e" => e
"123" => 123
"1 + 1" => 1 + 1
'now' => now
"12:30" => 12:30

# converters work on strings, in both forms
'hello' to upper => HELLO
upper('hello') => HELLO
"HeLLo" to lower => hello
lower("HeLLo") => hello

# strings do not mix with numbers
"abc" + 1 => !error Cannot add Number to Text
1 + "abc" => !error Cannot add Text to Number
2 * "abc" => !error Invalid operands: Number and Text
"abc" + "def" => !error Cannot add Text to Text

# a literal still being typed is committed as is
"unfinished => unfinished
'unfinished => unfinished
"trailing\ => trailing
47 changes: 47 additions & 0 deletions tests/string.cpp
Original file line number Diff line number Diff line change
@@ -0,0 +1,47 @@
#include "helpers.hpp"
#include "numen/numen.hpp"
#include <catch2/catch_test_macros.hpp>

TEST_CASE("empty string literals evaluate to empty text") {
test::assertExpr(R"("")", "");
test::assertExpr("''", "");
}

TEST_CASE("string literals preserve whitespace") {
test::assertExpr(R"(" a b ")", " a b ");
test::assertExpr("'\ttab'", "\ttab");
}

TEST_CASE("String literal alone should be marked as non converted") {
numen::Numen calc{};
auto res = calc.compute("'this is a string literal'");
REQUIRE(res);
REQUIRE(res->asStr());
REQUIRE(!res->conversion);
}

TEST_CASE("Transformed string should be marked as converted") {
numen::Numen calc{};
auto res = calc.compute("'this is a string literal' to upper");
REQUIRE(res);
REQUIRE(res->asStr());
REQUIRE(res->conversion);
}

TEST_CASE("Converter call form marks the conversion and names the converter") {
numen::Numen calc{};
auto res = calc.compute("upper('hello')");
REQUIRE(res);
REQUIRE(res->asStr());
CHECK(*res->asStr() == "HELLO");
REQUIRE(res->conversion);
REQUIRE(res->conversion->as<std::string>());
CHECK(res->conversion->as<std::string>()->to == "upper");
}

TEST_CASE("Regular functions do not mark a conversion") {
numen::Numen calc{};
auto res = calc.compute("sqrt(4)");
REQUIRE(res);
REQUIRE(!res->conversion);
}
Loading