diff --git a/CHANGELOG.md b/CHANGELOG.md index 05cf2c33..5956e2f9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,22 @@ All notable changes to this project are documented in this file. The format is based on Keep a Changelog, and this project adheres to Semantic Versioning. +## [Unreleased] + +### Added + +- Vertica dialect (`vertica`) across the Rust crate, FFI, Python, WASM, and + TypeScript SDK. Semantics follow the + [vertica-sqlglot-dialect](https://github.com/luisdelatorre012/vertica-sqlglot-dialect) + reference: 64-bit integer and double-precision type aliasing, `LONG VARCHAR` + / `LONG VARBINARY`, `MINUS`, the `//`, `!`, `!!`, and `@` operators, + `LISTAGG ... USING PARAMETERS`, native `NVL2`/`DECODE`/`ZEROIFNULL`, + `TIMESTAMPADD`/`DATEDIFF`, and statement-start `GETDATE()`/`SYSDATE`, which + lower to `STATEMENT_TIMESTAMP()` for PostgreSQL. Vertica's type-dependent + NULL ordering is never assumed when it is the source and is made explicit + when it is the target. Vertica-only DDL (projections, segmentation), + `TIMESERIES`, `MATCH`, and partitioned `LIMIT ... OVER` are not yet modeled. + ## [0.12.1] - 2026-09-21 ### Added diff --git a/Makefile b/Makefile index f1150570..f570d0b2 100644 --- a/Makefile +++ b/Makefile @@ -248,6 +248,7 @@ test-rust-feature-gates: cargo check -p polyglot-sql --no-default-features --features transpile,dialect-clickhouse,dialect-postgresql cargo check -p polyglot-sql --no-default-features --features transpile,dialect-tsql cargo check -p polyglot-sql --no-default-features --features transpile,dialect-fabric + cargo check -p polyglot-sql --no-default-features --features transpile,dialect-vertica cargo check -p polyglot-sql --no-default-features --features dialect-snowflake cargo check -p polyglot-sql --no-default-features --features generate,dialect-snowflake cargo check -p polyglot-sql --no-default-features --features transpile,dialect-snowflake diff --git a/README.md b/README.md index 12539a34..a9e626e0 100644 --- a/README.md +++ b/README.md @@ -37,7 +37,7 @@ Release notes are tracked in [`CHANGELOG.md`](CHANGELOG.md). | MySQL | Oracle | PostgreSQL | Presto | Redshift | | RisingWave | SingleStore | Snowflake | Solr | Spark | | SQLite | StarRocks | Tableau | Teradata | TiDB | -| Trino | TSQL | DataFusion | Generic SQL | | +| Trino | TSQL | Vertica | DataFusion | Generic SQL | ## Quick Start diff --git a/crates/polyglot-sql-ffi/src/dialects.rs b/crates/polyglot-sql-ffi/src/dialects.rs index 87331aa4..3329b150 100644 --- a/crates/polyglot-sql-ffi/src/dialects.rs +++ b/crates/polyglot-sql-ffi/src/dialects.rs @@ -3,7 +3,7 @@ use polyglot_sql::dialects::DialectType; use std::os::raw::c_char; use std::ptr; -const DIALECTS: [DialectType; 34] = [ +const DIALECTS: [DialectType; 35] = [ DialectType::Generic, DialectType::PostgreSQL, DialectType::MySQL, @@ -38,6 +38,7 @@ const DIALECTS: [DialectType; 34] = [ DialectType::Dremio, DialectType::Exasol, DialectType::DataFusion, + DialectType::Vertica, ]; /// Return supported dialect names as JSON. diff --git a/crates/polyglot-sql-ffi/tests/ffi_tests.rs b/crates/polyglot-sql-ffi/tests/ffi_tests.rs index 970c2935..6a2172ba 100644 --- a/crates/polyglot-sql-ffi/tests/ffi_tests.rs +++ b/crates/polyglot-sql-ffi/tests/ffi_tests.rs @@ -2050,7 +2050,7 @@ fn test_dialect_list_and_count() { let list: Vec = serde_json::from_str(&json).expect("invalid dialect list json"); let count = polyglot_dialect_count(); assert_eq!(list.len() as i32, count); - assert_eq!(count, 34); + assert_eq!(count, 35); let unique: BTreeSet<&str> = list.iter().map(String::as_str).collect(); assert_eq!(unique.len(), list.len()); assert!(list.iter().any(|d| d == "generic")); diff --git a/crates/polyglot-sql-python/README.md b/crates/polyglot-sql-python/README.md index 1a6c33aa..d8a7ad0a 100644 --- a/crates/polyglot-sql-python/README.md +++ b/crates/polyglot-sql-python/README.md @@ -329,7 +329,7 @@ All functions are exported from `polyglot_sql`. Current dialect names returned by `polyglot_sql.dialects()`: -`athena`, `bigquery`, `clickhouse`, `cockroachdb`, `datafusion`, `databricks`, `doris`, `dremio`, `drill`, `druid`, `duckdb`, `dune`, `exasol`, `fabric`, `generic`, `hive`, `materialize`, `mysql`, `oracle`, `postgres`, `presto`, `redshift`, `risingwave`, `singlestore`, `snowflake`, `solr`, `spark`, `sqlite`, `starrocks`, `tableau`, `teradata`, `tidb`, `trino`, `tsql`. +`athena`, `bigquery`, `clickhouse`, `cockroachdb`, `datafusion`, `databricks`, `doris`, `dremio`, `drill`, `druid`, `duckdb`, `dune`, `exasol`, `fabric`, `generic`, `hive`, `materialize`, `mysql`, `oracle`, `postgres`, `presto`, `redshift`, `risingwave`, `singlestore`, `snowflake`, `solr`, `spark`, `sqlite`, `starrocks`, `tableau`, `teradata`, `tidb`, `trino`, `tsql`, `vertica`. ## Error Handling diff --git a/crates/polyglot-sql-python/src/dialects.rs b/crates/polyglot-sql-python/src/dialects.rs index 028fbda3..7ee939c1 100644 --- a/crates/polyglot-sql-python/src/dialects.rs +++ b/crates/polyglot-sql-python/src/dialects.rs @@ -35,6 +35,7 @@ const DIALECT_NAMES: &[&str] = &[ "tidb", "trino", "tsql", + "vertica", ]; #[pyfunction] diff --git a/crates/polyglot-sql-python/tests/test_dialects.py b/crates/polyglot-sql-python/tests/test_dialects.py index 85cbfb48..a033d82f 100644 --- a/crates/polyglot-sql-python/tests/test_dialects.py +++ b/crates/polyglot-sql-python/tests/test_dialects.py @@ -36,6 +36,7 @@ "tidb", "trino", "tsql", + "vertica", } diff --git a/crates/polyglot-sql-wasm/Cargo.toml b/crates/polyglot-sql-wasm/Cargo.toml index 5669a150..b019a1c6 100644 --- a/crates/polyglot-sql-wasm/Cargo.toml +++ b/crates/polyglot-sql-wasm/Cargo.toml @@ -74,6 +74,7 @@ all-dialects = [ "dialect-druid", "dialect-solr", "dialect-tableau", "dialect-dune", "dialect-fabric", "dialect-drill", "dialect-dremio", "dialect-exasol", "dialect-datafusion", + "dialect-vertica", ] dialect-postgresql = ["polyglot-sql/dialect-postgresql"] dialect-mysql = ["polyglot-sql/dialect-mysql"] @@ -108,6 +109,7 @@ dialect-drill = ["polyglot-sql/dialect-drill"] dialect-dremio = ["polyglot-sql/dialect-dremio"] dialect-exasol = ["polyglot-sql/dialect-exasol"] dialect-datafusion = ["polyglot-sql/dialect-datafusion"] +dialect-vertica = ["polyglot-sql/dialect-vertica"] function-catalog-clickhouse = [ "dialect-clickhouse", "semantic", diff --git a/crates/polyglot-sql-wasm/src/lib.rs b/crates/polyglot-sql-wasm/src/lib.rs index 0e5e3e8e..8f793cac 100644 --- a/crates/polyglot-sql-wasm/src/lib.rs +++ b/crates/polyglot-sql-wasm/src/lib.rs @@ -851,6 +851,8 @@ fn get_dialects_internal() -> Vec<&'static str> { dialects.push("exasol"); #[cfg(feature = "dialect-datafusion")] dialects.push("datafusion"); + #[cfg(feature = "dialect-vertica")] + dialects.push("vertica"); dialects } @@ -2927,7 +2929,7 @@ mod tests { let dialects: Vec = serde_json::from_str(&result).unwrap(); let unique: std::collections::BTreeSet<&str> = dialects.iter().map(String::as_str).collect(); - assert_eq!(dialects.len(), 34); + assert_eq!(dialects.len(), 35); assert_eq!(unique.len(), dialects.len()); assert!(unique.contains("generic")); assert!(unique.contains("postgresql")); @@ -3570,6 +3572,7 @@ mod tests { "dremio", "exasol", "datafusion", + "vertica", ]; for dialect in dialects { diff --git a/crates/polyglot-sql/Cargo.toml b/crates/polyglot-sql/Cargo.toml index 29cceb09..15766266 100644 --- a/crates/polyglot-sql/Cargo.toml +++ b/crates/polyglot-sql/Cargo.toml @@ -48,6 +48,7 @@ all-dialects = [ "dialect-druid", "dialect-solr", "dialect-tableau", "dialect-dune", "dialect-fabric", "dialect-drill", "dialect-dremio", "dialect-exasol", "dialect-datafusion", + "dialect-vertica", ] dialect-postgresql = [] dialect-mysql = [] @@ -82,6 +83,7 @@ dialect-drill = [] dialect-dremio = [] dialect-exasol = [] dialect-datafusion = [] +dialect-vertica = [] function-catalog-clickhouse = [ "semantic", "dep:polyglot-sql-function-catalogs", diff --git a/crates/polyglot-sql/README.md b/crates/polyglot-sql/README.md index 348eeb48..c3d42a7f 100644 --- a/crates/polyglot-sql/README.md +++ b/crates/polyglot-sql/README.md @@ -488,7 +488,7 @@ assert_eq!(err.line(), None); ## Supported Dialects -Athena, BigQuery, ClickHouse, CockroachDB, DataFusion, Databricks, Doris, Dremio, Drill, Druid, DuckDB, Dune, Exasol, Fabric, Generic SQL, Hive, Materialize, MySQL, Oracle, PostgreSQL, Presto, Redshift, RisingWave, SingleStore, Snowflake, Solr, Spark, SQLite, StarRocks, Tableau, Teradata, TiDB, Trino, TSQL +Athena, BigQuery, ClickHouse, CockroachDB, DataFusion, Databricks, Doris, Dremio, Drill, Druid, DuckDB, Dune, Exasol, Fabric, Generic SQL, Hive, Materialize, MySQL, Oracle, PostgreSQL, Presto, Redshift, RisingWave, SingleStore, Snowflake, Solr, Spark, SQLite, StarRocks, Tableau, Teradata, TiDB, Trino, TSQL, Vertica ## Feature Flags diff --git a/crates/polyglot-sql/src/dialects/duckdb.rs b/crates/polyglot-sql/src/dialects/duckdb.rs index a49e5e78..6955f521 100644 --- a/crates/polyglot-sql/src/dialects/duckdb.rs +++ b/crates/polyglot-sql/src/dialects/duckdb.rs @@ -200,6 +200,7 @@ impl DialectImpl for DuckDBDialect { this: f.this, separator: f.separator, on_overflow: None, + max_length: None, order_by: f.order_by, distinct: f.distinct, filter: f.filter, @@ -216,6 +217,7 @@ impl DialectImpl for DuckDBDialect { this: f.this, separator: f.separator, on_overflow: None, + max_length: None, order_by: f.order_by, distinct: f.distinct, filter: f.filter, diff --git a/crates/polyglot-sql/src/dialects/exasol.rs b/crates/polyglot-sql/src/dialects/exasol.rs index 8360a095..067d1f3c 100644 --- a/crates/polyglot-sql/src/dialects/exasol.rs +++ b/crates/polyglot-sql/src/dialects/exasol.rs @@ -130,6 +130,7 @@ impl DialectImpl for ExasolDialect { this: f.this, separator: f.separator, on_overflow: None, + max_length: None, order_by: f.order_by, distinct: f.distinct, filter: f.filter, diff --git a/crates/polyglot-sql/src/dialects/mod.rs b/crates/polyglot-sql/src/dialects/mod.rs index 61cdc84c..66354877 100644 --- a/crates/polyglot-sql/src/dialects/mod.rs +++ b/crates/polyglot-sql/src/dialects/mod.rs @@ -89,6 +89,8 @@ mod tidb; mod trino; #[cfg(any(feature = "dialect-tsql", feature = "dialect-fabric"))] mod tsql; +#[cfg(feature = "dialect-vertica")] +mod vertica; pub use generic::GenericDialect; // Always available @@ -158,6 +160,8 @@ pub use tidb::TiDBDialect; pub use trino::TrinoDialect; #[cfg(feature = "dialect-tsql")] pub use tsql::TSQLDialect; +#[cfg(feature = "dialect-vertica")] +pub use vertica::VerticaDialect; use crate::error::Result; #[cfg(feature = "transpile")] @@ -274,6 +278,8 @@ pub enum DialectType { Exasol, /// Apache DataFusion -- Arrow-based query engine with modern SQL extensions. DataFusion, + /// Vertica (OpenText Analytics Database) -- columnar MPP analytic database. + Vertica, } impl DialectType { @@ -330,6 +336,7 @@ impl std::fmt::Display for DialectType { DialectType::Dremio => write!(f, "dremio"), DialectType::Exasol => write!(f, "exasol"), DialectType::DataFusion => write!(f, "datafusion"), + DialectType::Vertica => write!(f, "vertica"), } } } @@ -373,6 +380,7 @@ impl std::str::FromStr for DialectType { "dremio" => Ok(DialectType::Dremio), "exasol" => Ok(DialectType::Exasol), "datafusion" | "arrow-datafusion" | "arrow_datafusion" => Ok(DialectType::DataFusion), + "vertica" => Ok(DialectType::Vertica), _ => Err(crate::error::Error::parse( format!("Unknown dialect: {}", s), 0, @@ -2387,6 +2395,7 @@ cached_dialect!(CACHED_DRILL, DrillDialect, "dialect-drill"); cached_dialect!(CACHED_DREMIO, DremioDialect, "dialect-dremio"); cached_dialect!(CACHED_EXASOL, ExasolDialect, "dialect-exasol"); cached_dialect!(CACHED_DATAFUSION, DataFusionDialect, "dialect-datafusion"); +cached_dialect!(CACHED_VERTICA, VerticaDialect, "dialect-vertica"); fn configs_for_dialect_type(dt: DialectType) -> DialectConfigs { /// Clone configs from a cached static and pair with a fresh transform closure. @@ -2469,6 +2478,8 @@ fn configs_for_dialect_type(dt: DialectType) -> DialectConfigs { DialectType::Exasol => from_cache!(CACHED_EXASOL, ExasolDialect), #[cfg(feature = "dialect-datafusion")] DialectType::DataFusion => from_cache!(CACHED_DATAFUSION, DataFusionDialect), + #[cfg(feature = "dialect-vertica")] + DialectType::Vertica => from_cache!(CACHED_VERTICA, VerticaDialect), _ => from_cache!(CACHED_GENERIC, GenericDialect), } } @@ -3211,6 +3222,7 @@ impl Dialect { feature = "dialect-oracle", feature = "dialect-clickhouse", feature = "dialect-fabric", + feature = "dialect-vertica", ))] use crate::transforms; @@ -3380,6 +3392,13 @@ impl Dialect { // DataFusion supports QUALIFY and semi/anti joins natively #[cfg(feature = "dialect-datafusion")] DialectType::DataFusion => Ok(expr), + // Vertica doesn't support QUALIFY or semi/anti join syntax + #[cfg(feature = "dialect-vertica")] + DialectType::Vertica => { + let expr = transforms::eliminate_qualify(expr)?; + let expr = transforms::eliminate_semi_and_anti_joins(expr)?; + Ok(expr) + } // Oracle doesn't support QUALIFY #[cfg(feature = "dialect-oracle")] DialectType::Oracle => { diff --git a/crates/polyglot-sql/src/dialects/normalization/aggregates.rs b/crates/polyglot-sql/src/dialects/normalization/aggregates.rs index ed93f23a..a6b8a085 100644 --- a/crates/polyglot-sql/src/dialects/normalization/aggregates.rs +++ b/crates/polyglot-sql/src/dialects/normalization/aggregates.rs @@ -421,6 +421,7 @@ pub(super) fn rewrite( this: sa.this, separator: sa.separator, on_overflow: None, + max_length: None, order_by: sa.order_by, distinct: sa.distinct, filter: None, @@ -541,6 +542,7 @@ pub(super) fn rewrite( this, separator: Some(sep), on_overflow: None, + max_length: None, order_by: gc.order_by, distinct: gc.distinct, filter: gc.filter, @@ -634,6 +636,7 @@ pub(super) fn rewrite( this: gc.this, separator: Some(sep), on_overflow: None, + max_length: None, order_by: gc.order_by, distinct: gc.distinct, filter: None, diff --git a/crates/polyglot-sql/src/dialects/normalization/mod.rs b/crates/polyglot-sql/src/dialects/normalization/mod.rs index 1ca6eb0e..f3f63f55 100644 --- a/crates/polyglot-sql/src/dialects/normalization/mod.rs +++ b/crates/polyglot-sql/src/dialects/normalization/mod.rs @@ -15,6 +15,7 @@ mod scalar; mod statements; pub(in crate::dialects) mod temporal; mod types; +mod vertica; #[derive(Debug, Clone, Copy)] struct NormalizationContext { @@ -86,6 +87,13 @@ pub(super) fn normalize( let expr = statements::normalize_root(expr, &context); transform_recursive(expr, &|e| { + let e = if matches!(source, DialectType::Vertica) && !matches!(target, DialectType::Vertica) + { + vertica::normalize_from_vertica(e, target)? + } else { + e + }; + if matches!(source, DialectType::DataFusion) && matches!(target, DialectType::DuckDB) { if let Expression::Function(ref function) = e { if function.name.eq_ignore_ascii_case("NOW") && function.args.is_empty() { @@ -2363,15 +2371,18 @@ pub(super) fn normalize( | DialectType::Teradata | DialectType::Spark | DialectType::Databricks - | DialectType::Redshift => Action::None, + | DialectType::Redshift + | DialectType::Vertica => Action::None, _ => Action::Scalar(scalar::Action::Nvl2Expand), } } Expression::Decode(_) | Expression::DecodeCase(_) => { // DECODE(a, b, c[, d, e[, ...]]) -> CASE WHEN with null-safe comparisons - // Keep as DECODE for Oracle/Snowflake + // Keep as DECODE for Oracle/Snowflake/Vertica match target { - DialectType::Oracle | DialectType::Snowflake => Action::None, + DialectType::Oracle | DialectType::Snowflake | DialectType::Vertica => { + Action::None + } _ => Action::Scalar(scalar::Action::DecodeSimplify), } } @@ -2728,8 +2739,11 @@ pub(super) fn normalize( | DialectType::StarRocks | DialectType::Doris ); + // Vertica's implicit NULL placement depends on the sort key's data + // type (NULLS AUTO), so there is no source default to make explicit. if o.nulls_first.is_none() && source != target + && source != DialectType::Vertica && (target_supports_nulls || target_rewrites_nulls) { Action::Operators(operators::Action::NullsOrdering) diff --git a/crates/polyglot-sql/src/dialects/normalization/operators.rs b/crates/polyglot-sql/src/dialects/normalization/operators.rs index 2a95794c..0ba234c4 100644 --- a/crates/polyglot-sql/src/dialects/normalization/operators.rs +++ b/crates/polyglot-sql/src/dialects/normalization/operators.rs @@ -507,8 +507,11 @@ pub(super) fn rewrite( is_asc }; - // Only add explicit nulls ordering if source and target defaults differ - if source_nulls_first != target_nulls_first { + // Only add explicit nulls ordering if source and target defaults differ. + // Vertica's default is data-type dependent, so always make it explicit. + if source_nulls_first != target_nulls_first + || matches!(target, DialectType::Vertica) + { o.nulls_first = Some(source_nulls_first); } // If they match, leave nulls_first as None so the generator won't output it diff --git a/crates/polyglot-sql/src/dialects/normalization/scalar.rs b/crates/polyglot-sql/src/dialects/normalization/scalar.rs index ac339d69..2fe08cd8 100644 --- a/crates/polyglot-sql/src/dialects/normalization/scalar.rs +++ b/crates/polyglot-sql/src/dialects/normalization/scalar.rs @@ -5577,7 +5577,7 @@ pub(super) fn rewrite( )))) } } - DialectType::Redshift => { + DialectType::Redshift | DialectType::Vertica => { let unit = Expression::Identifier(Identifier::new("DAY")); Ok(Expression::Function(Box::new(Function::new( "DATEDIFF".to_string(), @@ -6830,9 +6830,12 @@ pub(super) fn rewrite( // GETDATE() -> CURRENT_TIMESTAMP for non-TSQL targets "GETDATE" if f.args.is_empty() => match target { DialectType::TSQL => Ok(Expression::Function(f)), - DialectType::Redshift => Ok(Expression::Function(Box::new( - Function::new("GETDATE".to_string(), vec![]), - ))), + DialectType::Redshift | DialectType::Vertica => { + Ok(Expression::Function(Box::new(Function::new( + "GETDATE".to_string(), + vec![], + )))) + } _ => Ok(Expression::CurrentTimestamp( crate::expressions::CurrentTimestamp { precision: None, @@ -7096,6 +7099,10 @@ pub(super) fn rewrite( DialectType::Oracle | DialectType::Redshift => { Ok(Expression::Function(f)) } + // Vertica: SYSDATE is a synonym for GETDATE() + DialectType::Vertica => Ok(Expression::Function(Box::new( + Function::new("GETDATE".to_string(), vec![]), + ))), DialectType::Snowflake => { // Snowflake uses SYSDATE() with parens let mut f = *f; @@ -10092,6 +10099,7 @@ pub(super) fn rewrite( | DialectType::Teradata | DialectType::Spark | DialectType::Databricks + | DialectType::Vertica ); if keep_as_decode { return Ok(Expression::Function(f)); @@ -11079,6 +11087,7 @@ pub(super) fn rewrite( this, separator, on_overflow: None, + max_length: None, order_by: None, distinct: false, filter: None, @@ -11674,6 +11683,7 @@ pub(super) fn rewrite( | DialectType::Teradata | DialectType::Spark | DialectType::Databricks + | DialectType::Vertica ); let (a, b, c) = if let Expression::Nvl2(nvl2) = e { if nvl2_native { diff --git a/crates/polyglot-sql/src/dialects/normalization/vertica.rs b/crates/polyglot-sql/src/dialects/normalization/vertica.rs new file mode 100644 index 00000000..c5e5f41e --- /dev/null +++ b/crates/polyglot-sql/src/dialects/normalization/vertica.rs @@ -0,0 +1,236 @@ +//! Rewrites for Vertica-specific source semantics. +//! +//! Vertica shares most of its surface syntax with PostgreSQL, but a few functions +//! carry Vertica-only meaning that must be lowered before a foreign target sees them. + +use super::*; +use crate::expressions::{ + AggFunc, AtTimeZone, DateAddFunc, GroupConcatFunc, Interval, IntervalUnit, IntervalUnitSpec, + StringAggFunc, VarArgFunc, +}; + +/// Rewrite a single node parsed as Vertica for a non-Vertica target. +pub(super) fn normalize_from_vertica(e: Expression, target: DialectType) -> Result { + match e { + Expression::Function(f) if f.args.is_empty() && !f.quoted => { + match f.name.to_ascii_uppercase().as_str() { + // GETDATE() and SYSDATE are the statement-start local timestamp, not the + // transaction-start CURRENT_TIMESTAMP. + "GETDATE" | "SYSDATE" => { + Ok(statement_timestamp(target, false).unwrap_or(Expression::Function(f))) + } + "GETUTCDATE" => { + Ok(statement_timestamp(target, true).unwrap_or(Expression::Function(f))) + } + _ => Ok(Expression::Function(f)), + } + } + Expression::CurrentTimestamp(ts) if ts.sysdate => { + Ok(statement_timestamp(target, false).unwrap_or(Expression::CurrentTimestamp(ts))) + } + Expression::Function(f) if !f.quoted => { + match (f.name.to_ascii_uppercase().as_str(), f.args.len()) { + // ZEROIFNULL(x) -> COALESCE(x, 0), except where it is native + ("ZEROIFNULL", 1) if !matches!(target, DialectType::Snowflake) => { + let mut args = f.args; + args.push(Expression::number(0)); + Ok(Expression::Coalesce(Box::new(VarArgFunc { + original_name: None, + expressions: args, + inferred_type: None, + }))) + } + // TIMESTAMPADD(unit, n, ts) -> the portable date-add node + ("TIMESTAMPADD", 3) => match timestamp_unit(&f.args[0]) { + Some(unit) => { + let mut args = f.args; + let this = args.pop().unwrap(); + let interval = args.pop().unwrap(); + if is_postgres_family(target) { + return Ok(postgres_interval_add(this, interval, unit)); + } + Ok(Expression::DateAdd(Box::new(DateAddFunc { + this, + interval, + unit, + }))) + } + None => Ok(Expression::Function(f)), + }, + // APPROXIMATE_COUNT_DISTINCT(x) -> the portable approximate-distinct node + ("APPROXIMATE_COUNT_DISTINCT", 1) => { + let this = f.args.into_iter().next().unwrap(); + Ok(Expression::ApproxDistinct(Box::new(AggFunc { + this, + distinct: false, + filter: None, + order_by: Vec::new(), + name: None, + ignore_nulls: None, + having_max: None, + limit: None, + inferred_type: None, + }))) + } + _ => Ok(Expression::Function(f)), + } + } + Expression::ListAgg(f) => Ok(lower_listagg(*f, target)), + // LISTAGG(...) WITHIN GROUP (ORDER BY ...): fold the ordering into the aggregate + // when the target's native form carries it inline. + Expression::WithinGroup(wg) if matches!(wg.this, Expression::ListAgg(_)) => { + let crate::expressions::WithinGroup { this, order_by } = *wg; + let Expression::ListAgg(mut f) = this else { + unreachable!() + }; + match lower_listagg(*f.clone(), target) { + Expression::ListAgg(lowered) => Ok(Expression::WithinGroup(Box::new( + crate::expressions::WithinGroup { + this: Expression::ListAgg(lowered), + order_by, + }, + ))), + _ => { + f.order_by = Some(order_by); + Ok(lower_listagg(*f, target)) + } + } + } + other => Ok(other), + } +} + +/// Vertica statement-start timestamps, for targets that can express them. +fn statement_timestamp(target: DialectType, utc: bool) -> Option { + let timestamp = |this: Expression| { + Expression::Cast(Box::new(Cast { + this, + to: DataType::Timestamp { + precision: None, + timezone: false, + }, + trailing_comments: Vec::new(), + double_colon_syntax: false, + format: None, + default: None, + inferred_type: None, + })) + }; + let at_utc = |this: Expression| { + Expression::AtTimeZone(Box::new(AtTimeZone { + this, + zone: Expression::string("UTC"), + })) + }; + match target { + // PostgreSQL has a true statement-start clock + DialectType::PostgreSQL => { + let now = Expression::Function(Box::new(Function::new( + "STATEMENT_TIMESTAMP".to_string(), + vec![], + ))); + Some(timestamp(if utc { at_utc(now) } else { now })) + } + // These targets have the same statement-start functions natively + DialectType::TSQL | DialectType::Fabric => None, + DialectType::Redshift if !utc => None, + _ if utc => Some(timestamp(at_utc(Expression::CurrentTimestamp( + crate::expressions::CurrentTimestamp { + precision: None, + sysdate: false, + }, + )))), + _ => None, + } +} + +/// LISTAGG defaults to a ',' separator in Vertica; spell it out and pick the +/// target's native string-aggregation form. +fn lower_listagg(mut f: crate::expressions::ListAggFunc, target: DialectType) -> Expression { + if f.separator.is_none() { + f.separator = Some(Expression::string(",")); + } + match target { + DialectType::PostgreSQL + | DialectType::Materialize + | DialectType::RisingWave + | DialectType::CockroachDB + | DialectType::TSQL + | DialectType::Fabric + | DialectType::BigQuery => Expression::StringAgg(Box::new(StringAggFunc { + this: f.this, + separator: f.separator, + order_by: f.order_by, + distinct: f.distinct, + filter: f.filter, + limit: None, + inferred_type: None, + })), + DialectType::MySQL + | DialectType::TiDB + | DialectType::SingleStore + | DialectType::Doris + | DialectType::StarRocks + | DialectType::SQLite => Expression::GroupConcat(Box::new(GroupConcatFunc { + this: f.this, + separator: f.separator, + order_by: f.order_by, + distinct: f.distinct, + filter: f.filter, + limit: None, + inferred_type: None, + })), + _ => Expression::ListAgg(Box::new(f)), + } +} + +/// Vertica datetime units accepted by TIMESTAMPADD, including the ODBC SQL_TSI_ forms. +fn timestamp_unit(unit: &Expression) -> Option { + let name = temporal::get_unit_str_static(unit); + Some(match name.trim_start_matches("SQL_TSI_") { + "YEAR" | "YEARS" | "YY" | "YYYY" => IntervalUnit::Year, + "QUARTER" | "QUARTERS" | "QQ" | "Q" => IntervalUnit::Quarter, + "MONTH" | "MONTHS" | "MM" | "M" => IntervalUnit::Month, + "WEEK" | "WEEKS" | "WK" | "WW" => IntervalUnit::Week, + "DAY" | "DAYS" | "DD" | "D" | "DAYOFYEAR" | "DY" | "Y" => IntervalUnit::Day, + "HOUR" | "HOURS" | "HH" => IntervalUnit::Hour, + "MINUTE" | "MINUTES" | "MI" | "N" => IntervalUnit::Minute, + "SECOND" | "SECONDS" | "SS" | "S" => IntervalUnit::Second, + "MILLISECOND" | "MILLISECONDS" | "MS" => IntervalUnit::Millisecond, + "MICROSECOND" | "MICROSECONDS" | "US" => IntervalUnit::Microsecond, + _ => return None, + }) +} + +fn is_postgres_family(target: DialectType) -> bool { + matches!( + target, + DialectType::PostgreSQL + | DialectType::Materialize + | DialectType::RisingWave + | DialectType::CockroachDB + ) +} + +/// `ts + INTERVAL 'n UNIT'` for literal amounts, `ts + INTERVAL '1 UNIT' * n` otherwise. +fn postgres_interval_add(ts: Expression, amount: Expression, unit: IntervalUnit) -> Expression { + let interval = |value: String| { + Expression::Interval(Box::new(Interval { + this: Some(Expression::string(&value)), + unit: Some(IntervalUnitSpec::Simple { + unit, + use_plural: false, + }), + })) + }; + let offset = match amount { + Expression::Literal(ref lit) if matches!(lit.as_ref(), Literal::Number(_)) => { + let Literal::Number(n) = lit.as_ref() else { + unreachable!() + }; + interval(n.clone()) + } + other => Expression::Mul(Box::new(BinaryOp::new(interval("1".to_string()), other))), + }; + Expression::Add(Box::new(BinaryOp::new(ts, offset))) +} diff --git a/crates/polyglot-sql/src/dialects/snowflake.rs b/crates/polyglot-sql/src/dialects/snowflake.rs index 7deabe77..3553a419 100644 --- a/crates/polyglot-sql/src/dialects/snowflake.rs +++ b/crates/polyglot-sql/src/dialects/snowflake.rs @@ -203,6 +203,7 @@ impl DialectImpl for SnowflakeDialect { this: f.this, separator: f.separator, on_overflow: None, + max_length: None, order_by: f.order_by, distinct: f.distinct, filter: f.filter, diff --git a/crates/polyglot-sql/src/dialects/vertica.rs b/crates/polyglot-sql/src/dialects/vertica.rs new file mode 100644 index 00000000..6de3213b --- /dev/null +++ b/crates/polyglot-sql/src/dialects/vertica.rs @@ -0,0 +1,294 @@ +//! Vertica Dialect +//! +//! Vertica (OpenText Analytics Database) is a columnar MPP analytic database whose +//! SQL surface is largely PostgreSQL-compatible. +//! Reference: https://docs.vertica.com/latest/en/sql-reference/ +//! Semantics cross-checked against https://github.com/luisdelatorre012/vertica-sqlglot-dialect +//! +//! Key characteristics: +//! - Double-quote identifiers, case-insensitive (folded to lowercase) +//! - `::` casts, ILIKE, `||` concatenation, LIMIT/OFFSET +//! - MINUS is an alias for EXCEPT +//! - All integer types are 64-bit (INT, SMALLINT, TINYINT are BIGINT) +//! - REAL / FLOAT are DOUBLE PRECISION +//! - LONG VARCHAR / LONG VARBINARY types +//! - NVL, NVL2, DECODE, ZEROIFNULL, LISTAGG, DATEDIFF(unit, a, b), TIMESTAMPADD +//! - GETDATE() / SYSDATE are statement-start timestamps +//! - No QUALIFY, TRY_CAST or semi/anti join syntax +//! - No nested comments + +use super::{DialectImpl, DialectType}; +#[cfg(feature = "transpile")] +use crate::error::Result; +#[cfg(feature = "transpile")] +use crate::expressions::{ + AggFunc, Case, Expression, Function, Identifier, IntervalUnit, ListAggFunc, Literal, VarArgFunc, +}; +#[cfg(feature = "generate")] +use crate::generator::GeneratorConfig; +use crate::tokens::TokenizerConfig; + +/// Vertica dialect +pub struct VerticaDialect; + +impl DialectImpl for VerticaDialect { + fn dialect_type(&self) -> DialectType { + DialectType::Vertica + } + + fn tokenizer_config(&self) -> TokenizerConfig { + use crate::tokens::TokenType; + let mut config = TokenizerConfig::default(); + // Vertica uses double quotes for identifiers (PostgreSQL-style) + config.identifiers.insert('"', '"'); + // Vertica does NOT support nested comments + config.nested_comments = false; + // `//` is integer division + config.double_slash_int_div = true; + // MINUS is an alias for EXCEPT in Vertica + config + .keywords + .insert("MINUS".to_string(), TokenType::Except); + config + } + + #[cfg(feature = "generate")] + fn generator_config(&self) -> GeneratorConfig { + use crate::generator::{IdentifierQuoteStyle, LimitFetchStyle}; + GeneratorConfig { + identifier_quote: '"', + identifier_quote_style: IdentifierQuoteStyle::DOUBLE_QUOTE, + dialect: Some(DialectType::Vertica), + single_string_interval: true, + locking_reads_supported: false, + limit_fetch_style: LimitFetchStyle::Limit, + nvl2_supported: true, + supports_median: true, + ..Default::default() + } + } + + #[cfg(feature = "transpile")] + fn transform_expr(&self, expr: Expression) -> Result { + match expr { + // IFNULL -> COALESCE in Vertica + Expression::IfNull(f) => Ok(Expression::Coalesce(Box::new(VarArgFunc { + original_name: None, + expressions: vec![f.this, f.expression], + inferred_type: None, + }))), + + // Coalesce with original_name (e.g., IFNULL parsed as Coalesce) -> clear original_name + Expression::Coalesce(mut f) => { + f.original_name = None; + Ok(Expression::Coalesce(f)) + } + + // Vertica has no TRY_CAST; fall back to CAST + Expression::TryCast(c) => Ok(Expression::Cast(c)), + Expression::SafeCast(c) => Ok(Expression::Cast(c)), + + // CountIf -> SUM(CASE WHEN condition THEN 1 ELSE 0 END) + Expression::CountIf(f) => { + let case_expr = Expression::Case(Box::new(Case { + operand: None, + whens: vec![(f.this.clone(), Expression::number(1))], + else_: Some(Expression::number(0)), + comments: Vec::new(), + inferred_type: None, + })); + Ok(Expression::Sum(Box::new(AggFunc { + ignore_nulls: None, + having_max: None, + this: case_expr, + distinct: f.distinct, + filter: f.filter, + order_by: Vec::new(), + name: None, + limit: None, + inferred_type: None, + }))) + } + + // RAND -> RANDOM in Vertica + Expression::Rand(r) => { + let _ = r.seed; + Ok(Expression::Random(crate::expressions::Random)) + } + + // DAYOFWEEK_ISO has no shared generator form + Expression::DayOfWeekIso(f) => Ok(Expression::Function(Box::new(Function::new( + "DAYOFWEEK_ISO".to_string(), + vec![f.this], + )))), + + // DATE_ADD / DATEADD -> TIMESTAMPADD(unit, n, ts); Vertica has no DATEADD + Expression::DateAdd(f) => Ok(timestamp_add( + interval_unit_name(&f.unit), + f.interval, + f.this, + )), + + // APPROX_COUNT_DISTINCT -> APPROXIMATE_COUNT_DISTINCT + Expression::ApproxDistinct(f) | Expression::ApproxCountDistinct(f) => { + Ok(Expression::Function(Box::new(Function::new( + "APPROXIMATE_COUNT_DISTINCT".to_string(), + vec![f.this], + )))) + } + + // GROUP_CONCAT / STRING_AGG -> LISTAGG + Expression::GroupConcat(f) => Ok(Expression::ListAgg(Box::new(ListAggFunc { + this: f.this, + separator: f.separator, + on_overflow: None, + max_length: None, + order_by: f.order_by, + distinct: f.distinct, + filter: f.filter, + inferred_type: None, + }))), + Expression::StringAgg(f) => Ok(Expression::ListAgg(Box::new(ListAggFunc { + this: f.this, + separator: f.separator, + on_overflow: None, + max_length: None, + order_by: f.order_by, + distinct: f.distinct, + filter: f.filter, + inferred_type: None, + }))), + + // Generic function transformations + Expression::Function(f) => self.transform_function(*f), + + // Pass through everything else + _ => Ok(expr), + } + } +} + +#[cfg(feature = "transpile")] +impl VerticaDialect { + fn transform_function(&self, f: Function) -> Result { + let name_upper = f.name.to_uppercase(); + match name_upper.as_str() { + // IFNULL / ISNULL -> COALESCE + "IFNULL" | "ISNULL" if f.args.len() == 2 => { + Ok(Expression::Coalesce(Box::new(VarArgFunc { + original_name: None, + expressions: f.args, + inferred_type: None, + }))) + } + + // SYSDATE is a synonym for GETDATE (statement-start timestamp) + "SYSDATE" if f.args.is_empty() => Ok(Expression::Function(Box::new(Function::new( + "GETDATE".to_string(), + vec![], + )))), + + // TIMESTAMPDIFF(unit, a, b) is a synonym for DATEDIFF(unit, a, b) + "TIMESTAMPDIFF" if f.args.len() == 3 => { + let mut args = f.args; + upper_unit(&mut args[0]); + Ok(Expression::Function(Box::new(Function::new( + "DATEDIFF".to_string(), + args, + )))) + } + + // APPROX_COUNT_DISTINCT -> APPROXIMATE_COUNT_DISTINCT + "APPROX_COUNT_DISTINCT" if !f.args.is_empty() => Ok(Expression::Function(Box::new( + Function::new("APPROXIMATE_COUNT_DISTINCT".to_string(), f.args), + ))), + + // DATEADD(unit, n, ts) -> TIMESTAMPADD(unit, n, ts) + "DATEADD" | "DATE_ADD" if f.args.len() == 3 => { + let mut args = f.args; + let ts = args.pop().unwrap(); + let n = args.pop().unwrap(); + let unit = args.pop().unwrap(); + let unit = match unit { + Expression::Literal(lit) => match *lit { + Literal::String(s) => { + Expression::Identifier(Identifier::new(s.to_ascii_uppercase())) + } + other => Expression::Literal(Box::new(other)), + }, + other => other, + }; + Ok(Expression::Function(Box::new(Function::new( + "TIMESTAMPADD".to_string(), + vec![unit, n, ts], + )))) + } + + // TIMESTAMPADD(unit, n, ts): normalize the unit keyword to upper case + "TIMESTAMPADD" if f.args.len() == 3 => { + let mut f = f; + upper_unit(&mut f.args[0]); + Ok(Expression::Function(Box::new(f))) + } + + // CHARINDEX(substr, str[, start]) -> INSTR(str, substr[, start]) + "CHARINDEX" if f.args.len() >= 2 => { + let mut args = f.args; + let substr = args.remove(0); + let string = args.remove(0); + let mut new_args = vec![string, substr]; + new_args.extend(args); + Ok(Expression::Function(Box::new(Function::new( + "INSTR".to_string(), + new_args, + )))) + } + + // LEN -> LENGTH + "LEN" if f.args.len() == 1 => Ok(Expression::Function(Box::new(Function::new( + "LENGTH".to_string(), + f.args, + )))), + + // Pass through everything else + _ => Ok(Expression::Function(Box::new(f))), + } + } +} + +#[cfg(feature = "transpile")] +fn interval_unit_name(unit: &IntervalUnit) -> &'static str { + match unit { + IntervalUnit::Year => "YEAR", + IntervalUnit::Quarter => "QUARTER", + IntervalUnit::Month => "MONTH", + IntervalUnit::Week => "WEEK", + IntervalUnit::Day => "DAY", + IntervalUnit::Hour => "HOUR", + IntervalUnit::Minute => "MINUTE", + IntervalUnit::Second => "SECOND", + IntervalUnit::Millisecond => "MILLISECOND", + IntervalUnit::Microsecond => "MICROSECOND", + IntervalUnit::Nanosecond => "NANOSECOND", + } +} + +#[cfg(feature = "transpile")] +fn timestamp_add(unit: &str, amount: Expression, ts: Expression) -> Expression { + Expression::Function(Box::new(Function::new( + "TIMESTAMPADD".to_string(), + vec![Expression::Identifier(Identifier::new(unit)), amount, ts], + ))) +} + +/// Upper-case a bare datetime unit keyword (`day` -> `DAY`), leaving expressions alone. +#[cfg(feature = "transpile")] +fn upper_unit(unit: &mut Expression) { + match unit { + Expression::Identifier(id) if !id.quoted => id.name = id.name.to_ascii_uppercase(), + Expression::Column(col) if col.table.is_none() && !col.name.quoted => { + *unit = Expression::Identifier(Identifier::new(col.name.name.to_ascii_uppercase())); + } + _ => {} + } +} diff --git a/crates/polyglot-sql/src/expressions.rs b/crates/polyglot-sql/src/expressions.rs index c88b6b6b..66093257 100644 --- a/crates/polyglot-sql/src/expressions.rs +++ b/crates/polyglot-sql/src/expressions.rs @@ -6811,6 +6811,9 @@ pub struct ListAggFunc { pub this: Expression, pub separator: Option, pub on_overflow: Option, + /// Vertica `USING PARAMETERS max_length = n` + #[serde(default, skip_serializing_if = "Option::is_none")] + pub max_length: Option>, pub order_by: Option>, pub distinct: bool, pub filter: Option, diff --git a/crates/polyglot-sql/src/generator.rs b/crates/polyglot-sql/src/generator.rs index cfb57ee1..d9b06e95 100644 --- a/crates/polyglot-sql/src/generator.rs +++ b/crates/polyglot-sql/src/generator.rs @@ -3248,8 +3248,11 @@ impl Generator { self.generate_expression(&f.expression)?; self.write(")"); Ok(()) - } else if matches!(self.config.dialect, Some(DialectType::DuckDB)) { - // DuckDB uses // operator for integer division + } else if matches!( + self.config.dialect, + Some(DialectType::DuckDB) | Some(DialectType::Vertica) + ) { + // DuckDB and Vertica use // operator for integer division self.generate_expression(&f.this)?; self.write(" // "); self.generate_expression(&f.expression)?; @@ -18583,6 +18586,32 @@ impl Generator { } fn generate_function(&mut self, func: &Function) -> Result<()> { + // Vertica spells factorial as the postfix `!` operator + if self.config.dialect == Some(DialectType::Vertica) + && func.name.eq_ignore_ascii_case("FACTORIAL") + && func.args.len() == 1 + && !func.quoted + { + let operand = &func.args[0]; + let atomic = matches!( + operand, + Expression::Literal(_) + | Expression::Column(_) + | Expression::Identifier(_) + | Expression::Paren(_) + | Expression::Function(_) + ); + if !atomic { + self.write("("); + } + self.generate_expression(operand)?; + if !atomic { + self.write(")"); + } + self.write("!"); + return Ok(()); + } + // Normalize function name based on dialect settings let normalized_name = if func.name.eq_ignore_ascii_case("GROUPING") && func.args.len() > 1 @@ -20969,6 +20998,12 @@ impl Generator { self.write("()"); return Ok(()); } + Some(DialectType::Vertica) => { + // Vertica SYSDATE is a synonym for GETDATE() + self.write_keyword("GETDATE"); + self.write("()"); + return Ok(()); + } _ => { // Other dialects use CURRENT_TIMESTAMP for SYSDATE } @@ -21279,8 +21314,11 @@ impl Generator { fn generate_if_func(&mut self, f: &IfFunc) -> Result<()> { use crate::dialects::DialectType; - // Generic mode: normalize IF to CASE WHEN - if self.config.dialect.is_none() || self.config.dialect == Some(DialectType::Generic) { + // Generic mode and dialects without an IF function: normalize IF to CASE WHEN + if matches!( + self.config.dialect, + None | Some(DialectType::Generic) | Some(DialectType::Vertica) + ) { self.write_keyword("CASE WHEN"); self.write_space(); self.generate_expression(&f.condition)?; @@ -21769,6 +21807,43 @@ impl Generator { ) } + /// Vertica passes LISTAGG options as `USING PARAMETERS name = value, ...`. + fn generate_vertica_listagg_parameters(&mut self, f: &ListAggFunc) -> Result<()> { + let mut params: Vec<(&str, Option<&Expression>, Option<&str>)> = Vec::new(); + if let Some(ref sep) = f.separator { + params.push(("separator", Some(sep), None)); + } + if let Some(ref max_length) = f.max_length { + params.push(("max_length", Some(max_length), None)); + } + match f.on_overflow { + Some(ListAggOverflow::Error) => params.push(("on_overflow", None, Some("'ERROR'"))), + Some(ListAggOverflow::Truncate { .. }) => { + params.push(("on_overflow", None, Some("'TRUNCATE'"))) + } + None => {} + } + if params.is_empty() { + return Ok(()); + } + self.write_space(); + self.write_keyword("USING PARAMETERS"); + self.write_space(); + for (i, (name, value, literal)) in params.into_iter().enumerate() { + if i > 0 { + self.write(", "); + } + self.write(name); + self.write(" = "); + match (value, literal) { + (Some(value), _) => self.generate_expression(value)?, + (None, Some(literal)) => self.write(literal), + (None, None) => {} + } + } + Ok(()) + } + fn generate_listagg(&mut self, f: &ListAggFunc) -> Result<()> { use crate::dialects::DialectType; let order_inside_args = matches!(self.config.dialect, Some(DialectType::DuckDB)); @@ -21779,7 +21854,12 @@ impl Generator { self.write_space(); } self.generate_expression(&f.this)?; - if let Some(ref sep) = f.separator { + if f.max_length.is_some() && self.config.dialect != Some(DialectType::Vertica) { + self.unsupported("LISTAGG max_length is not supported in this dialect")?; + } + if self.config.dialect == Some(DialectType::Vertica) { + self.generate_vertica_listagg_parameters(f)?; + } else if let Some(ref sep) = f.separator { self.write(", "); self.generate_expression(sep)?; } else if matches!( @@ -21789,7 +21869,11 @@ impl Generator { // Trino/Presto require explicit separator; default to ',' self.write(", ','"); } - if let Some(ref overflow) = f.on_overflow { + if let Some(ref overflow) = f + .on_overflow + .as_ref() + .filter(|_| self.config.dialect != Some(DialectType::Vertica)) + { self.write_space(); self.write_keyword("ON OVERFLOW"); self.write_space(); @@ -25918,9 +26002,55 @@ impl Generator { Ok(()) } + /// Vertica type spellings that differ from the shared defaults. + /// + /// Every Vertica integer is 64-bit and every float is an 8-byte double, so + /// narrower types widen instead of being silently truncated. Returns false + /// when the shared generator should handle the type. + fn generate_vertica_data_type(&mut self, dt: &DataType) -> bool { + match dt { + DataType::TinyInt { .. } + | DataType::SmallInt { .. } + | DataType::Int { .. } + | DataType::BigInt { .. } => self.write_keyword("BIGINT"), + DataType::Float { .. } | DataType::Double { .. } => { + self.write_keyword("DOUBLE PRECISION") + } + DataType::Text => self.write_keyword("LONG VARCHAR"), + DataType::String { length: None } => self.write_keyword("VARCHAR"), + DataType::TextWithLength { length } => { + self.write_keyword("LONG VARCHAR"); + self.write(&format!("({})", length)); + } + DataType::String { + length: Some(length), + } => { + self.write_keyword("VARCHAR"); + self.write(&format!("({})", length)); + } + DataType::Blob => self.write_keyword("LONG VARBINARY"), + DataType::Time { + precision, + timezone: true, + } => { + self.write_keyword("TIMETZ"); + if let Some(p) = precision { + self.write(&format!("({})", p)); + } + } + _ => return false, + } + true + } + fn generate_data_type(&mut self, dt: &DataType) -> Result<()> { use crate::dialects::DialectType; + if self.config.dialect == Some(DialectType::Vertica) && self.generate_vertica_data_type(dt) + { + return Ok(()); + } + match dt { DataType::Boolean => { // Dialect-specific boolean type mappings @@ -37685,6 +37815,7 @@ impl Generator { | Some(DialectType::Oracle) | Some(DialectType::BigQuery) | Some(DialectType::Teradata) + | Some(DialectType::Vertica) ) { self.write_keyword("INSTR"); self.write("("); diff --git a/crates/polyglot-sql/src/optimizer/set_operation_types.rs b/crates/polyglot-sql/src/optimizer/set_operation_types.rs index 45f1f095..894f4ece 100644 --- a/crates/polyglot-sql/src/optimizer/set_operation_types.rs +++ b/crates/polyglot-sql/src/optimizer/set_operation_types.rs @@ -328,7 +328,7 @@ fn family(dialect: DialectType) -> Family { DataFusion => Family::Arrow, Doris | StarRocks => Family::Standard, Drill | Dremio => Family::Limited, - Exasol => Family::Standard, + Exasol | Vertica => Family::Standard, Druid | Solr | Tableau => Family::Limited, } } @@ -680,6 +680,7 @@ fn decimal_common(l: &DataType, r: &DataType, dialect: DialectType) -> Option 65, DialectType::Exasol => 36, + DialectType::Vertica => 1024, DialectType::ClickHouse => 76, _ => 38, }; diff --git a/crates/polyglot-sql/src/parser.rs b/crates/polyglot-sql/src/parser.rs index 2bab9243..7f0403bf 100644 --- a/crates/polyglot-sql/src/parser.rs +++ b/crates/polyglot-sql/src/parser.rs @@ -30965,8 +30965,22 @@ impl Parser { /// Parse unary expressions fn parse_unary(&mut self) -> Result { let mut prefixes = Vec::new(); + let is_vertica = self.config.dialect == Some(crate::dialects::DialectType::Vertica); while !self.is_at_end() { let token = self.peek().token_type; + // Vertica: `@ x` is ABS(x) and `!! x` is the prefix factorial + if is_vertica + && (token == TokenType::DAt + || (token == TokenType::Exclamation && self.check_next(TokenType::Exclamation))) + { + let _scope = self.enter_parser_depth(prefixes.len() + 1)?; + if token == TokenType::Exclamation { + self.skip(); + } + prefixes.push(token); + self.skip(); + continue; + } if !matches!( token, TokenType::Plus @@ -31002,6 +31016,8 @@ impl Parser { TokenType::PipeSlash => { Expression::Sqrt(Box::new(UnaryFunc::with_name(expr, "|/".to_string()))) } + TokenType::DAt => Expression::Abs(Box::new(UnaryFunc::new(expr))), + TokenType::Exclamation => factorial(expr), _ => unreachable!("collected prefix operator"), }; } @@ -31171,6 +31187,14 @@ impl Parser { } } + // Vertica: postfix `x!` is factorial + if self.config.dialect == Some(crate::dialects::DialectType::Vertica) { + while self.check(TokenType::Exclamation) && !self.check_next(TokenType::Exclamation) { + self.skip(); + expr = factorial(expr); + } + } + // Handle EXCLAMATION for Snowflake model attribute syntax: model!PREDICT(...) while self.match_token(TokenType::Exclamation) { // Parse the attribute/function after the exclamation mark @@ -38076,13 +38100,63 @@ impl Parser { // Check for optional DISTINCT let distinct = self.match_token(TokenType::Distinct); let this = self.parse_expression()?; - let separator = if self.match_token(TokenType::Comma) { + let mut separator = if self.match_token(TokenType::Comma) { Some(self.parse_expression()?) } else { None }; + let mut max_length = None; + let mut vertica_overflow = None; + // Vertica: LISTAGG(expr USING PARAMETERS separator = ',', max_length = n, + // on_overflow = 'ERROR' | 'TRUNCATE') + if self.config.dialect == Some(crate::dialects::DialectType::Vertica) + && self.check(TokenType::Using) + && self.check_next_identifier("PARAMETERS") + { + self.skip(); + self.skip(); + loop { + let name = self.expect_identifier_or_keyword()?; + self.expect(TokenType::Eq)?; + let value = self.parse_primary()?; + match name.to_ascii_lowercase().as_str() { + "separator" => separator = Some(value), + "max_length" => max_length = Some(Box::new(value)), + "on_overflow" => { + let mode = match &value { + Expression::Literal(lit) => match lit.as_ref() { + Literal::String(mode) => mode.to_ascii_uppercase(), + _ => String::new(), + }, + _ => String::new(), + }; + vertica_overflow = Some(match mode.as_str() { + "ERROR" => ListAggOverflow::Error, + "TRUNCATE" => ListAggOverflow::Truncate { + filler: None, + with_count: false, + }, + _ => { + return Err(self.parse_error( + "LISTAGG on_overflow must be 'ERROR' or 'TRUNCATE'", + )) + } + }); + } + other => { + return Err(self + .parse_error(format!("Unknown LISTAGG parameter: {}", other))) + } + } + if !self.match_token(TokenType::Comma) { + break; + } + } + } // Parse optional ON OVERFLOW clause - let on_overflow = if self.match_token(TokenType::On) { + let on_overflow = if vertica_overflow.is_some() { + vertica_overflow + } else if self.match_token(TokenType::On) { if self.match_identifier("OVERFLOW") { if self.match_identifier("ERROR") { Some(ListAggOverflow::Error) @@ -38119,6 +38193,7 @@ impl Parser { this, separator, on_overflow, + max_length, order_by: None, distinct, filter: None, @@ -42312,9 +42387,56 @@ impl Parser { && name.eq_ignore_ascii_case("LARGEINT")) } + /// Vertica interval qualifiers may carry a seconds precision: `SECOND(3)`. + fn parse_interval_field_precision(&mut self, unit: String) -> Result { + if self.config.dialect == Some(crate::dialects::DialectType::Vertica) + && self.check(TokenType::LParen) + && self.check_next(TokenType::Number) + { + self.skip(); + let precision = self.expect_number()?; + self.expect(TokenType::RParen)?; + return Ok(format!("{}({})", unit, precision)); + } + Ok(unit) + } + /// Parse a data type. fn parse_data_type(&mut self) -> Result { - self.with_parser_depth(|parser| parser.parse_data_type_inner()) + let data_type = self.with_parser_depth(|parser| parser.parse_data_type_inner())?; + Ok(self.normalize_dialect_data_type(data_type)) + } + + /// Apply dialect-specific type aliasing after a data type has been parsed. + /// + /// Vertica stores every integer type as a signed 64-bit integer and every + /// floating-point type as an 8-byte double, so the narrower spellings are + /// aliases rather than distinct types. + fn normalize_dialect_data_type(&self, data_type: DataType) -> DataType { + if self.config.dialect != Some(crate::dialects::DialectType::Vertica) { + return data_type; + } + match data_type { + DataType::Int { .. } | DataType::SmallInt { .. } | DataType::TinyInt { .. } => { + DataType::BigInt { length: None } + } + DataType::Float { .. } => DataType::Double { + precision: None, + scale: None, + }, + DataType::Custom { ref name } + if name.eq_ignore_ascii_case("INT8") || name.eq_ignore_ascii_case("INT4") => + { + DataType::BigInt { length: None } + } + DataType::Custom { ref name } if name.eq_ignore_ascii_case("FLOAT8") => { + DataType::Double { + precision: None, + scale: None, + } + } + other => other, + } } #[inline(never)] @@ -42875,7 +42997,10 @@ impl Parser { && !self.check(TokenType::RParen) && !self.check(TokenType::Comma) { - Some(self.advance_text()?.to_ascii_uppercase()) + { + let unit = self.advance_text()?.to_ascii_uppercase(); + Some(self.parse_interval_field_precision(unit)?) + } } else { None }; @@ -42885,7 +43010,10 @@ impl Parser { || self.check(TokenType::Var) || self.check_keyword() { - Some(self.advance_text()?.to_ascii_uppercase()) + { + let unit = self.advance_text()?.to_ascii_uppercase(); + Some(self.parse_interval_field_precision(unit)?) + } } else { None } @@ -43225,10 +43353,27 @@ impl Parser { }) } } - // LONG VARCHAR (Exasol) - same as TEXT + // LONG VARCHAR (Exasol, Vertica) - same as TEXT + // LONG VARBINARY (Vertica) - same as BLOB "LONG" => { if self.match_identifier("VARCHAR") { - Ok(DataType::Text) + if self.match_token(TokenType::LParen) { + let length = self.expect_number()? as u32; + self.expect(TokenType::RParen)?; + Ok(DataType::TextWithLength { length }) + } else { + Ok(DataType::Text) + } + } else if self.match_identifier("VARBINARY") { + if self.match_token(TokenType::LParen) { + let length = self.expect_number()?; + self.expect(TokenType::RParen)?; + Ok(DataType::Custom { + name: format!("LONG VARBINARY({})", length), + }) + } else { + Ok(DataType::Blob) + } } else { Ok(DataType::Custom { name: "LONG".to_string(), @@ -43535,7 +43680,8 @@ impl Parser { /// For other dialects (like Snowflake), brackets are subscript operations /// (e.g., x::VARIANT[0] means cast to VARIANT, then subscript with [0]). fn parse_data_type_for_cast(&mut self) -> Result { - self.with_parser_depth(|parser| parser.parse_data_type_for_cast_inner()) + let data_type = self.with_parser_depth(|parser| parser.parse_data_type_for_cast_inner())?; + Ok(self.normalize_dialect_data_type(data_type)) } #[inline(never)] @@ -43881,7 +44027,10 @@ impl Parser { && !self.check(TokenType::Not) && !self.check(TokenType::Null) { - Some(self.advance_text()?.to_ascii_uppercase()) + { + let unit = self.advance_text()?.to_ascii_uppercase(); + Some(self.parse_interval_field_precision(unit)?) + } } else { None }; @@ -43891,7 +44040,10 @@ impl Parser { || self.check(TokenType::Var) || self.check_keyword() { - Some(self.advance_text()?.to_ascii_uppercase()) + { + let unit = self.advance_text()?.to_ascii_uppercase(); + Some(self.parse_interval_field_precision(unit)?) + } } else { None } @@ -67653,3 +67805,8 @@ mod explicit_eof_token_tests { ); } } + +/// FACTORIAL(x), produced by Vertica's `x!` and `!! x` operators. +fn factorial(expr: Expression) -> Expression { + Expression::Function(Box::new(Function::new("FACTORIAL".to_string(), vec![expr]))) +} diff --git a/crates/polyglot-sql/src/tokens.rs b/crates/polyglot-sql/src/tokens.rs index 9d78bde8..24a0ae9f 100644 --- a/crates/polyglot-sql/src/tokens.rs +++ b/crates/polyglot-sql/src/tokens.rs @@ -1551,6 +1551,8 @@ pub struct TokenizerConfig { /// end-of-input as the close. This is only enabled for ClickHouse fixture /// coverage, where some extracted corpus rows contain partial string probes. pub recover_unterminated_string: bool, + /// Whether `//` is the integer-division operator (Vertica). + pub double_slash_int_div: bool, } impl Default for TokenizerConfig { @@ -1584,6 +1586,7 @@ impl Default for TokenizerConfig { numbers_can_be_underscore_separated: false, recover_terminal_backslash_quote: false, recover_unterminated_string: false, + double_slash_int_div: false, } } } @@ -2644,6 +2647,7 @@ impl<'a, C: TokenizerCursor, T: TokenOutput> TokenizerState<'a, C, T> { ('>', '>') => Some(TokenType::GtGt), ('|', '|') => Some(TokenType::DPipe), ('|', '/') => Some(TokenType::PipeSlash), // Square root - PostgreSQL + ('/', '/') if self.config.double_slash_int_div => Some(TokenType::Div), // Vertica (':', ':') => Some(TokenType::DColon), (':', '=') => Some(TokenType::ColonEq), // := (assignment, named args) (':', '>') => Some(TokenType::ColonGt), // ::> (TSQL) diff --git a/crates/polyglot-sql/tests/common/test_runner.rs b/crates/polyglot-sql/tests/common/test_runner.rs index 1263ce8c..43a898cd 100644 --- a/crates/polyglot-sql/tests/common/test_runner.rs +++ b/crates/polyglot-sql/tests/common/test_runner.rs @@ -310,6 +310,7 @@ pub fn parse_dialect(name: &str) -> Option { "fabric" => Some(DialectType::Fabric), "solr" => Some(DialectType::Solr), "datafusion" | "arrow-datafusion" | "arrow_datafusion" => Some(DialectType::DataFusion), + "vertica" => Some(DialectType::Vertica), _ => None, } } diff --git a/crates/polyglot-sql/tests/custom_fixtures/vertica/identity.json b/crates/polyglot-sql/tests/custom_fixtures/vertica/identity.json new file mode 100644 index 00000000..f81f856e --- /dev/null +++ b/crates/polyglot-sql/tests/custom_fixtures/vertica/identity.json @@ -0,0 +1,214 @@ +{ + "dialect": "vertica", + "category": "identity", + "identity": [ + { + "sql": "SELECT 1", + "description": "Simple literal select" + }, + { + "sql": "SELECT a, b FROM t WHERE a > 1 ORDER BY b LIMIT 10 OFFSET 5", + "description": "SELECT with LIMIT/OFFSET" + }, + { + "sql": "SELECT \"Mixed Case\" FROM \"My Table\"", + "description": "Double-quoted identifiers" + }, + { + "sql": "SELECT CAST(a AS VARCHAR(10)) FROM t", + "description": "CAST" + }, + { + "sql": "SELECT a::VARCHAR(10) FROM t", + "expected": "SELECT CAST(a AS VARCHAR(10)) FROM t", + "description": "Double-colon cast" + }, + { + "sql": "SELECT * FROM t WHERE name ILIKE 'a%'", + "description": "ILIKE" + }, + { + "sql": "SELECT * FROM t WHERE name NOT ILIKE 'a%'", + "description": "NOT ILIKE" + }, + { + "sql": "SELECT a || b FROM t", + "description": "String concatenation" + }, + { + "sql": "SELECT NVL(a, b) FROM t", + "description": "NVL is native" + }, + { + "sql": "SELECT NVL2(a, b, c) FROM t", + "description": "NVL2 is native" + }, + { + "sql": "SELECT DECODE(a, 1, 'one', 2, 'two', 'other') FROM t", + "description": "DECODE is native" + }, + { + "sql": "SELECT ZEROIFNULL(a) FROM t", + "description": "ZEROIFNULL is native" + }, + { + "sql": "SELECT COALESCE(a, b, c) FROM t", + "description": "COALESCE" + }, + { + "sql": "SELECT DATEDIFF(DAY, started_at, ended_at) FROM events", + "description": "DATEDIFF with unit" + }, + { + "sql": "SELECT DATEDIFF('day', started_at, ended_at) FROM events", + "description": "DATEDIFF with string unit" + }, + { + "sql": "SELECT TIMESTAMPADD(day, 3, occurred_at) FROM events", + "expected": "SELECT TIMESTAMPADD(DAY, 3, occurred_at) FROM events", + "description": "TIMESTAMPADD unit is upper-cased" + }, + { + "sql": "SELECT TIMESTAMPDIFF(hour, started_at, ended_at) FROM events", + "expected": "SELECT DATEDIFF(HOUR, started_at, ended_at) FROM events", + "description": "TIMESTAMPDIFF is DATEDIFF" + }, + { + "sql": "SELECT ADD_MONTHS(invoice_date, 2) FROM invoices", + "description": "ADD_MONTHS" + }, + { + "sql": "SELECT GETDATE(), GETUTCDATE(), CURRENT_TIMESTAMP(3)", + "description": "Statement and transaction timestamps" + }, + { + "sql": "SELECT SYSDATE, SYSDATE()", + "expected": "SELECT GETDATE(), GETDATE()", + "description": "SYSDATE is a synonym for GETDATE" + }, + { + "sql": "SELECT TIME_SLICE(ts, 5, 'MINUTE', 'START') FROM ticks", + "description": "TIME_SLICE" + }, + { + "sql": "SELECT DAYOFMONTH(ts), DAYOFWEEK(ts), DAYOFWEEK_ISO(ts), DAYOFYEAR(ts) FROM t", + "description": "DAYOF* functions" + }, + { + "sql": "SELECT TO_CHAR(ts, 'YYYY-MM-DD'), TO_DATE('2020-01-01', 'YYYY-MM-DD') FROM t", + "description": "TO_CHAR/TO_DATE" + }, + { + "sql": "SELECT REGEXP_LIKE(a, 'x.*') FROM t", + "description": "REGEXP_LIKE" + }, + { + "sql": "SELECT APPROXIMATE_COUNT_DISTINCT(a) FROM t", + "description": "APPROXIMATE_COUNT_DISTINCT" + }, + { + "sql": "SELECT MEDIAN(b) OVER (PARTITION BY c) FROM t", + "description": "MEDIAN analytic" + }, + { + "sql": "SELECT ROW_NUMBER() OVER (PARTITION BY k ORDER BY a DESC) FROM t", + "description": "Window function" + }, + { + "sql": "SELECT LISTAGG(city) FROM places", + "description": "LISTAGG default separator" + }, + { + "sql": "SELECT LISTAGG(DISTINCT city) FROM places", + "description": "LISTAGG DISTINCT" + }, + { + "sql": "SELECT LISTAGG(name, ',') WITHIN GROUP (ORDER BY ordinal) FROM names", + "expected": "SELECT LISTAGG(name USING PARAMETERS separator = ',') WITHIN GROUP (ORDER BY ordinal) FROM names", + "description": "Two-argument LISTAGG becomes USING PARAMETERS" + }, + { + "sql": "SELECT LISTAGG(city USING PARAMETERS separator=' | ', max_length=4096, on_overflow='TRUNCATE') WITHIN GROUP (ORDER BY city) FROM places", + "expected": "SELECT LISTAGG(city USING PARAMETERS separator = ' | ', max_length = 4096, on_overflow = 'TRUNCATE') WITHIN GROUP (ORDER BY city) FROM places", + "description": "LISTAGG USING PARAMETERS" + }, + { + "sql": "SELECT a FROM t ORDER BY a NULLS FIRST, b DESC NULLS LAST", + "description": "Explicit NULL placement" + }, + { + "sql": "SELECT 1 MINUS SELECT 2", + "expected": "SELECT 1 EXCEPT SELECT 2", + "description": "MINUS is EXCEPT" + }, + { + "sql": "SELECT 1 UNION ALL SELECT 2 INTERSECT SELECT 3", + "description": "Set operation chain" + }, + { + "sql": "SELECT 117.32 // 2.5, !! 5, 4.98!, @ -5.0", + "expected": "SELECT 117.32 // 2.5, 5!, 4.98!, ABS(-5.0)", + "description": "Integer division, factorial and absolute value operators" + }, + { + "sql": "SELECT -4!, (-4)!, 5! + 1", + "description": "Factorial precedence" + }, + { + "sql": "SELECT |/ 25.0, ||/ 27.0", + "expected": "SELECT SQRT(25.0), CBRT(27.0)", + "description": "Square and cube root operators" + }, + { + "sql": "SELECT a != b FROM t", + "expected": "SELECT a <> b FROM t", + "description": "Not-equal is not factorial" + }, + { + "sql": "SELECT INTERVAL '1 year 2 months' YEAR TO MONTH", + "description": "Interval YEAR TO MONTH" + }, + { + "sql": "SELECT INTERVAL '3 days' DAY TO SECOND(3)", + "description": "Interval with seconds precision" + }, + { + "sql": "SELECT INTERVAL '1.234' SECOND(3)", + "description": "Interval SECOND precision" + }, + { + "sql": "SELECT ARRAY[1, 2], ROW(1, 'x')", + "description": "Collection constructors" + }, + { + "sql": "SELECT * FROM t QUALIFY ROW_NUMBER() OVER (PARTITION BY a ORDER BY b) = 1", + "expected": "SELECT * FROM (SELECT *, ROW_NUMBER() OVER (PARTITION BY a ORDER BY b) AS _w FROM t) AS _t WHERE _w = 1", + "description": "QUALIFY is rewritten to a subquery" + }, + { + "sql": "CREATE LOCAL TEMPORARY TABLE t (id INT) ON COMMIT DELETE ROWS", + "description": "Local temporary table" + }, + { + "sql": "CREATE LOCAL TEMP TABLE t_local ON COMMIT PRESERVE ROWS AS SELECT 1 AS c", + "description": "Local temporary CTAS" + }, + { + "sql": "INSERT INTO t (a, b) VALUES (1, 'x')", + "description": "INSERT" + }, + { + "sql": "UPDATE t SET a = 1 WHERE b = 2", + "description": "UPDATE" + }, + { + "sql": "DELETE FROM t WHERE a = 1", + "description": "DELETE" + }, + { + "sql": "MERGE INTO t USING s ON t.id = s.id WHEN MATCHED THEN UPDATE SET a = s.a WHEN NOT MATCHED THEN INSERT (id, a) VALUES (s.id, s.a)", + "description": "MERGE" + } + ], + "transpilation": [] +} diff --git a/crates/polyglot-sql/tests/custom_fixtures/vertica/transpilation.json b/crates/polyglot-sql/tests/custom_fixtures/vertica/transpilation.json new file mode 100644 index 00000000..477bc5f3 --- /dev/null +++ b/crates/polyglot-sql/tests/custom_fixtures/vertica/transpilation.json @@ -0,0 +1,248 @@ +{ + "dialect": "vertica", + "category": "transpilation", + "identity": [], + "transpilation": [ + { + "sql": "SELECT 1 EXCEPT SELECT 2", + "write": { + "postgresql": "SELECT 1 EXCEPT SELECT 2", + "duckdb": "SELECT 1 EXCEPT SELECT 2" + }, + "description": "MINUS/EXCEPT" + }, + { + "sql": "SELECT 1 FROM dual EXCEPT SELECT 2 FROM dual", + "read": { + "oracle": "SELECT 1 FROM dual MINUS SELECT 2 FROM dual" + }, + "description": "Oracle MINUS" + }, + { + "sql": "SELECT GETDATE(), GETUTCDATE()", + "write": { + "postgresql": "SELECT CAST(STATEMENT_TIMESTAMP() AS TIMESTAMP), CAST(STATEMENT_TIMESTAMP() AT TIME ZONE 'UTC' AS TIMESTAMP)", + "tsql": "SELECT GETDATE(), GETUTCDATE()", + "duckdb": "SELECT CURRENT_TIMESTAMP, CAST(CURRENT_TIMESTAMP AT TIME ZONE 'UTC' AS TIMESTAMP)" + }, + "description": "Statement-start timestamps lower to STATEMENT_TIMESTAMP in PostgreSQL" + }, + { + "sql": "SELECT GETDATE()", + "read": { + "redshift": "SELECT SYSDATE", + "tsql": "SELECT GETDATE()" + }, + "description": "SYSDATE / GETDATE into Vertica" + }, + { + "sql": "SELECT DATEDIFF(DAY, a, b) FROM t", + "write": { + "postgresql": "SELECT CAST(EXTRACT(epoch FROM CAST(b AS TIMESTAMP) - CAST(a AS TIMESTAMP)) / 86400 AS BIGINT) FROM t", + "snowflake": "SELECT DATEDIFF(DAY, a, b) FROM t", + "tsql": "SELECT DATEDIFF(DAY, a, b) FROM t", + "duckdb": "SELECT DATE_DIFF('DAY', a, b) FROM t" + }, + "read": { + "mysql": "SELECT DATEDIFF(b, a) FROM t", + "snowflake": "SELECT DATEDIFF(day, a, b) FROM t" + }, + "description": "DATEDIFF" + }, + { + "sql": "SELECT TIMESTAMPADD(DAY, 3, ts) FROM t", + "write": { + "postgresql": "SELECT ts + INTERVAL '3 DAY' FROM t", + "mysql": "SELECT DATE_ADD(ts, INTERVAL 3 DAY) FROM t", + "snowflake": "SELECT DATEADD(DAY, 3, ts) FROM t", + "tsql": "SELECT DATEADD(DAY, 3, ts) FROM t", + "duckdb": "SELECT ts + INTERVAL 3 DAY FROM t" + }, + "read": { + "snowflake": "SELECT DATEADD(day, 3, ts) FROM t", + "tsql": "SELECT DATEADD(day, 3, ts) FROM t" + }, + "description": "TIMESTAMPADD and DATEADD" + }, + { + "sql": "SELECT TIMESTAMPADD(MONTH, n, ts) FROM t", + "write": { + "postgresql": "SELECT ts + INTERVAL '1 MONTH' * n FROM t" + }, + "description": "TIMESTAMPADD with a non-literal amount" + }, + { + "sql": "SELECT LISTAGG(name USING PARAMETERS separator = ' | ') WITHIN GROUP (ORDER BY ordinal DESC) FROM names", + "write": { + "postgresql": "SELECT STRING_AGG(name, ' | ' ORDER BY ordinal DESC) FROM names", + "mysql": "SELECT GROUP_CONCAT(name ORDER BY ordinal DESC SEPARATOR ' | ') FROM names", + "snowflake": "SELECT LISTAGG(name, ' | ') WITHIN GROUP (ORDER BY ordinal DESC) FROM names", + "duckdb": "SELECT LISTAGG(name, ' | ' ORDER BY ordinal DESC) FROM names", + "tsql": "SELECT STRING_AGG(name, ' | ') WITHIN GROUP (ORDER BY ordinal DESC) FROM names" + }, + "description": "LISTAGG to native string aggregation" + }, + { + "sql": "SELECT LISTAGG(name USING PARAMETERS separator = ' | ') WITHIN GROUP (ORDER BY ordinal DESC NULLS FIRST) FROM names", + "read": { + "postgresql": "SELECT STRING_AGG(name, ' | ' ORDER BY ordinal DESC NULLS FIRST) FROM names" + }, + "description": "STRING_AGG to LISTAGG" + }, + { + "sql": "SELECT LISTAGG(name) FROM names", + "write": { + "postgresql": "SELECT STRING_AGG(name, ',') FROM names", + "snowflake": "SELECT LISTAGG(name, ',') FROM names" + }, + "description": "LISTAGG default separator is a comma" + }, + { + "sql": "SELECT LISTAGG(x USING PARAMETERS separator = ',') FROM t", + "read": { + "mysql": "SELECT GROUP_CONCAT(x SEPARATOR ',') FROM t" + }, + "description": "GROUP_CONCAT to LISTAGG" + }, + { + "sql": "SELECT NVL(a, b), NVL2(a, b, c), DECODE(a, 1, 'one', 'other'), ZEROIFNULL(a) FROM t", + "write": { + "postgresql": "SELECT COALESCE(a, b), CASE WHEN NOT a IS NULL THEN b ELSE c END, CASE WHEN a = 1 THEN 'one' ELSE 'other' END, COALESCE(a, 0) FROM t", + "snowflake": "SELECT COALESCE(a, b), NVL2(a, b, c), DECODE(a, 1, 'one', 'other'), IFF(a IS NULL, 0, a) FROM t", + "oracle": "SELECT NVL(a, b), NVL2(a, b, c), DECODE(a, 1, 'one', 'other'), NVL(a, 0) FROM t" + }, + "description": "NULL-handling functions" + }, + { + "sql": "SELECT DECODE(a, 1, 'x', 'y') FROM t", + "read": { + "oracle": "SELECT DECODE(a, 1, 'x', 'y') FROM t" + }, + "description": "DECODE stays native" + }, + { + "sql": "SELECT APPROXIMATE_COUNT_DISTINCT(a) FROM t", + "write": { + "snowflake": "SELECT APPROX_COUNT_DISTINCT(a) FROM t", + "duckdb": "SELECT APPROX_COUNT_DISTINCT(a) FROM t" + }, + "read": { + "snowflake": "SELECT APPROX_COUNT_DISTINCT(a) FROM t" + }, + "description": "Approximate distinct count" + }, + { + "sql": "SELECT COALESCE(a, b) FROM t", + "read": { + "mysql": "SELECT IFNULL(a, b) FROM t", + "tsql": "SELECT ISNULL(a, b) FROM t" + }, + "description": "IFNULL/ISNULL to COALESCE" + }, + { + "sql": "SELECT CAST(a AS BIGINT) FROM t", + "write": { + "postgresql": "SELECT CAST(a AS BIGINT) FROM t", + "tsql": "SELECT CAST(a AS BIGINT) FROM t" + }, + "read": { + "snowflake": "SELECT TRY_CAST(a AS INT) FROM t", + "postgresql": "SELECT a::INTEGER FROM t" + }, + "description": "Integers are BIGINT; no TRY_CAST" + }, + { + "sql": "SELECT CASE WHEN a > 1 THEN 'x' ELSE 'y' END FROM t", + "read": { + "snowflake": "SELECT IFF(a > 1, 'x', 'y') FROM t", + "mysql": "SELECT IF(a > 1, 'x', 'y') FROM t" + }, + "description": "IF/IFF to CASE" + }, + { + "sql": "SELECT INSTR(a, 'x') FROM t", + "read": { + "tsql": "SELECT CHARINDEX('x', a) FROM t" + }, + "description": "CHARINDEX to INSTR" + }, + { + "sql": "SELECT SUM(CASE WHEN a > 1 THEN 1 ELSE 0 END) FROM t", + "read": { + "snowflake": "SELECT COUNT_IF(a > 1) FROM t" + }, + "description": "COUNT_IF expansion" + }, + { + "sql": "SELECT * FROM a WHERE EXISTS((SELECT 1 FROM b WHERE a.id = b.id))", + "read": { + "spark": "SELECT * FROM a LEFT SEMI JOIN b ON a.id = b.id" + }, + "description": "Semi join elimination" + }, + { + "sql": "SELECT a FROM t LIMIT 5", + "write": { + "tsql": "SELECT TOP 5 a FROM t" + }, + "read": { + "tsql": "SELECT TOP 5 a FROM t" + }, + "description": "TOP to LIMIT" + }, + { + "sql": "SELECT a FROM t ORDER BY a", + "write": { + "postgresql": "SELECT a FROM t ORDER BY a", + "duckdb": "SELECT a FROM t ORDER BY a", + "mysql": "SELECT a FROM t ORDER BY a" + }, + "description": "Vertica NULL ordering is type-dependent, so none is assumed" + }, + { + "sql": "SELECT a FROM t ORDER BY a NULLS LAST", + "read": { + "postgresql": "SELECT a FROM t ORDER BY a" + }, + "description": "Source NULL ordering is made explicit" + }, + { + "sql": "SELECT a FROM t ORDER BY a DESC NULLS LAST", + "read": { + "duckdb": "SELECT a FROM t ORDER BY a DESC" + }, + "description": "DuckDB NULLS LAST default is made explicit" + }, + { + "sql": "SELECT 7 // 2, 5!, ABS(a) FROM t", + "write": { + "postgresql": "SELECT DIV(7, 2), FACTORIAL(5), ABS(a) FROM t", + "duckdb": "SELECT 7 // 2, FACTORIAL(5), ABS(a) FROM t" + }, + "description": "Vertica-only operators" + }, + { + "sql": "CREATE TABLE t (a BIGINT, b DOUBLE PRECISION, c LONG VARCHAR, d LONG VARBINARY, e VARBINARY, f TIMESTAMPTZ)", + "write": { + "postgresql": "CREATE TABLE t (a BIGINT, b DOUBLE PRECISION, c TEXT, d BYTEA, e BYTEA, f TIMESTAMPTZ)", + "duckdb": "CREATE TABLE t (a BIGINT, b DOUBLE, c TEXT, d VARBINARY, e BLOB, f TIMESTAMPTZ)", + "snowflake": "CREATE TABLE t (a BIGINT, b DOUBLE, c VARCHAR, d BLOB, e VARBINARY, f TIMESTAMPTZ)" + }, + "description": "Type mapping" + }, + { + "sql": "CREATE TABLE t (a BIGINT, b DOUBLE PRECISION, c LONG VARCHAR, d VARBINARY, e TIMESTAMPTZ)", + "read": { + "postgresql": "CREATE TABLE t (a INT, b REAL, c TEXT, d BYTEA, e TIMESTAMPTZ)" + }, + "description": "PostgreSQL types widen to Vertica types" + }, + { + "sql": "SELECT CAST(a AS BIGINT), CAST(b AS VARCHAR), CAST(c AS DOUBLE PRECISION), CAST(d AS VARBINARY) FROM t", + "read": { + "bigquery": "SELECT SAFE_CAST(a AS INT64), CAST(b AS STRING), CAST(c AS FLOAT64), CAST(d AS BYTES) FROM t" + }, + "description": "BigQuery types" + } + ] +} diff --git a/crates/polyglot-sql/tests/custom_fixtures/vertica/types.json b/crates/polyglot-sql/tests/custom_fixtures/vertica/types.json new file mode 100644 index 00000000..d7925adb --- /dev/null +++ b/crates/polyglot-sql/tests/custom_fixtures/vertica/types.json @@ -0,0 +1,50 @@ +{ + "dialect": "vertica", + "category": "types", + "identity": [ + { + "sql": "CREATE TABLE t (a INT, b INTEGER, c SMALLINT, d TINYINT, e INT8, f BIGINT)", + "expected": "CREATE TABLE t (a BIGINT, b BIGINT, c BIGINT, d BIGINT, e BIGINT, f BIGINT)", + "description": "All integers are 64-bit BIGINT" + }, + { + "sql": "CREATE TABLE t (a REAL, b FLOAT, c FLOAT8, d DOUBLE PRECISION)", + "expected": "CREATE TABLE t (a DOUBLE PRECISION, b DOUBLE PRECISION, c DOUBLE PRECISION, d DOUBLE PRECISION)", + "description": "All floats are DOUBLE PRECISION" + }, + { + "sql": "CREATE TABLE t (a LONG VARCHAR, b LONG VARCHAR(100000), c LONG VARBINARY, d LONG VARBINARY(200))", + "description": "LONG types" + }, + { + "sql": "CREATE TABLE t (a BINARY VARYING(32), b VARBINARY(10), c BINARY(4), d BYTEA)", + "expected": "CREATE TABLE t (a VARBINARY(32), b VARBINARY(10), c BINARY(4), d VARBINARY)", + "description": "Binary types" + }, + { + "sql": "CREATE TABLE t (a VARCHAR(80), b CHAR(3), c TEXT)", + "expected": "CREATE TABLE t (a VARCHAR(80), b CHAR(3), c LONG VARCHAR)", + "description": "Character types; TEXT is LONG VARCHAR" + }, + { + "sql": "CREATE TABLE t (a NUMERIC(10, 2), b DECIMAL(38, 0), c BOOLEAN, d UUID, e DATE)", + "expected": "CREATE TABLE t (a DECIMAL(10, 2), b DECIMAL(38, 0), c BOOLEAN, d UUID, e DATE)", + "description": "Numeric and misc types" + }, + { + "sql": "CREATE TABLE t (a TIMESTAMP, b TIMESTAMPTZ, c TIME, d TIMETZ, e TIMESTAMP WITH TIME ZONE)", + "expected": "CREATE TABLE t (a TIMESTAMP, b TIMESTAMPTZ, c TIME, d TIMETZ, e TIMESTAMPTZ)", + "description": "Temporal types" + }, + { + "sql": "CREATE TABLE t (a INTERVAL, b INTERVAL SECOND(3), c INTERVAL DAY TO SECOND(5), d INTERVAL YEAR TO MONTH)", + "description": "Interval column types" + }, + { + "sql": "SELECT CAST(a AS INT), a::FLOAT, CAST(b AS INTERVAL DAY TO SECOND(3)) FROM t", + "expected": "SELECT CAST(a AS BIGINT), CAST(a AS DOUBLE PRECISION), CAST(b AS INTERVAL DAY TO SECOND(3)) FROM t", + "description": "Cast target types" + } + ], + "transpilation": [] +} diff --git a/docs/set-operation-types.md b/docs/set-operation-types.md index ad13300a..2d998c36 100644 --- a/docs/set-operation-types.md +++ b/docs/set-operation-types.md @@ -40,7 +40,7 @@ every SQL expression is executable on a particular engine. ## Dialect rules and evidence -The policy covers all 33 named dialects and Generic. It is intentionally a +The policy covers all 34 named dialects and Generic. It is intentionally a partial type system: unsupported combinations return no hint, rather than claiming a complete implementation of an engine's implicit-cast rules. @@ -62,7 +62,7 @@ claiming a complete implementation of an engine's implicit-cast rules. | ClickHouse | Numeric representability, Nullable propagation, and compatible nested types. Int64/Float64 and Int64/UInt64 do not silently become lossy common types. | | Teradata | Compatible known operands retain the first SELECT's type, as specified by the engine. | | DataFusion | Numeric and decimal coercion, string/numeric union output, recursive nested types, and STRUCT matching by field name. Unsupported Arrow-specific types are unresolved. | -| Doris, StarRocks, Exasol, Generic | Same-category scalar widening and explicit decimal shapes. Engine-specific complex and cross-category coercions remain unresolved. | +| Doris, StarRocks, Exasol, Vertica, Generic | Same-category scalar widening and explicit decimal shapes. Engine-specific complex and cross-category coercions remain unresolved. | | Drill, Dremio, Druid, Solr, Tableau | Conservative scalar rules; mixed floating, decimal, temporal, and complex combinations without a supported rule remain unresolved. Druid DECIMAL/REAL normalize to DOUBLE. | Engine documentation and source used to distinguish these rules: diff --git a/packages/documentation/README.md b/packages/documentation/README.md index 9cbfa359..23958569 100644 --- a/packages/documentation/README.md +++ b/packages/documentation/README.md @@ -80,7 +80,7 @@ Default guard values: `maxInputBytes=16 MiB`, `maxTokens=1_000_000`, `maxAstNode ## Supported Dialects -Athena, BigQuery, ClickHouse, CockroachDB, DataFusion, Databricks, Doris, Dremio, Drill, Druid, DuckDB, Dune, Exasol, Fabric, Generic SQL, Hive, Materialize, MySQL, Oracle, PostgreSQL, Presto, Redshift, RisingWave, SingleStore, Snowflake, Solr, Spark, SQLite, StarRocks, Tableau, Teradata, TiDB, Trino, and TSQL (SQL Server). +Athena, BigQuery, ClickHouse, CockroachDB, DataFusion, Databricks, Doris, Dremio, Drill, Druid, DuckDB, Dune, Exasol, Fabric, Generic SQL, Hive, Materialize, MySQL, Oracle, PostgreSQL, Presto, Redshift, RisingWave, SingleStore, Snowflake, Solr, Spark, SQLite, StarRocks, Tableau, Teradata, TiDB, Trino, TSQL (SQL Server), and Vertica. ## Links diff --git a/packages/playground/src/lib/constants.ts b/packages/playground/src/lib/constants.ts index a2cf9587..4f8ba250 100644 --- a/packages/playground/src/lib/constants.ts +++ b/packages/playground/src/lib/constants.ts @@ -33,6 +33,7 @@ export const DIALECT_DISPLAY_NAMES: Record = { tidb: "TiDB", trino: "Trino", tsql: "SQL Server (T-SQL)", + vertica: "Vertica", }; export const DEFAULT_TRANSPILE_SQL = `SELECT diff --git a/packages/sdk/README.md b/packages/sdk/README.md index cd98d04e..92d9e2ba 100644 --- a/packages/sdk/README.md +++ b/packages/sdk/README.md @@ -1060,6 +1060,7 @@ This limit covers recursive parsing, not arbitrary programmatic AST construction | TiDB | `Dialect.TiDB` | | Trino | `Dialect.Trino` | | TSQL | `Dialect.TSQL` | +| Vertica | `Dialect.Vertica` | ## CDN Usage diff --git a/packages/sdk/src/index.ts b/packages/sdk/src/index.ts index 9d556f85..98695a55 100644 --- a/packages/sdk/src/index.ts +++ b/packages/sdk/src/index.ts @@ -49,6 +49,7 @@ export enum Dialect { Dremio = 'dremio', Exasol = 'exasol', DataFusion = 'datafusion', + Vertica = 'vertica', } /**