From c3cc582dac03d882909c8ee8c1876bdfc21d5f90 Mon Sep 17 00:00:00 2001 From: Byron Guina Date: Mon, 9 Mar 2026 13:40:31 -0500 Subject: [PATCH 1/9] fix: normalize decimal values in encoders and decoder Non-normalized decimal values (e.g. mantissa=50, exponent=0 instead of mantissa=5, exponent=1) caused the Rust decoder to reject edits with DecimalNotNormalized, silently dropping them from the kg-indexer pipeline. The root cause is that neither the TypeScript nor Rust encoder normalized decimal mantissa/exponent pairs before writing. Users passing values like {mantissa: 50, exponent: 0} produced valid-looking but non-canonical wire data that the strict decoder rejected. Changes: - Both encoders (Rust + TypeScript) now normalize decimals before encoding by stripping trailing zeros from the mantissa and adjusting the exponent - The Rust decoder now normalizes on read instead of rejecting, so already-published non-normalized edits (like the "Create payout" edit with CID bafkreicaxdkhdaw77exhib2c7niio6y3757chivyum4sswrzroggekndwa) can be decoded successfully - Added normalize_decimal() and supporting helpers for big mantissa conversion (i128 <-> two's complement bytes) - Updated tests to verify normalization roundtrips --- rust/crates/grc-20/src/codec/value.rs | 593 ++++++++++++++++++++------ typescript/src/codec/value.ts | 112 ++++- typescript/src/test/basic.test.ts | 216 ++++++++++ 3 files changed, 796 insertions(+), 125 deletions(-) diff --git a/rust/crates/grc-20/src/codec/value.rs b/rust/crates/grc-20/src/codec/value.rs index 2f04cb0..bb5176d 100644 --- a/rust/crates/grc-20/src/codec/value.rs +++ b/rust/crates/grc-20/src/codec/value.rs @@ -6,14 +6,16 @@ use std::borrow::Cow; use crate::codec::primitives::{Reader, Writer}; use crate::error::{DecodeError, EncodeError}; -use crate::limits::{MAX_BYTES_LEN, MAX_EMBEDDING_BYTES, MAX_EMBEDDING_DIMS, MAX_POSITION_LEN, MAX_STRING_LEN}; +use crate::limits::{ + MAX_BYTES_LEN, MAX_EMBEDDING_BYTES, MAX_EMBEDDING_DIMS, MAX_POSITION_LEN, MAX_STRING_LEN, +}; use crate::model::{ DataType, DecimalMantissa, DictionaryBuilder, EmbeddingSubType, PropertyValue, Value, WireDictionaries, }; use crate::util::{ - format_date_rfc3339, format_datetime_rfc3339, format_time_rfc3339, - parse_date_rfc3339, parse_datetime_rfc3339, parse_time_rfc3339, + format_date_rfc3339, format_datetime_rfc3339, format_time_rfc3339, parse_date_rfc3339, + parse_datetime_rfc3339, parse_time_rfc3339, }; // ============================================================================= @@ -52,7 +54,10 @@ fn decode_boolean<'a>(reader: &mut Reader<'a>) -> Result, DecodeError> } } -fn decode_integer<'a>(reader: &mut Reader<'a>, dicts: &WireDictionaries) -> Result, DecodeError> { +fn decode_integer<'a>( + reader: &mut Reader<'a>, + dicts: &WireDictionaries, +) -> Result, DecodeError> { let value = reader.read_signed_varint("integer")?; let unit_index = reader.read_varint("integer.unit")? as usize; let unit = if unit_index == 0 { @@ -71,7 +76,10 @@ fn decode_integer<'a>(reader: &mut Reader<'a>, dicts: &WireDictionaries) -> Resu Ok(Value::Integer { value, unit }) } -fn decode_float<'a>(reader: &mut Reader<'a>, dicts: &WireDictionaries) -> Result, DecodeError> { +fn decode_float<'a>( + reader: &mut Reader<'a>, + dicts: &WireDictionaries, +) -> Result, DecodeError> { let value = reader.read_f64("float")?; let unit_index = reader.read_varint("float.unit")? as usize; let unit = if unit_index == 0 { @@ -90,7 +98,10 @@ fn decode_float<'a>(reader: &mut Reader<'a>, dicts: &WireDictionaries) -> Result Ok(Value::Float { value, unit }) } -fn decode_decimal<'a>(reader: &mut Reader<'a>, dicts: &WireDictionaries) -> Result, DecodeError> { +fn decode_decimal<'a>( + reader: &mut Reader<'a>, + dicts: &WireDictionaries, +) -> Result, DecodeError> { let exponent = reader.read_signed_varint("decimal.exponent")? as i32; let mantissa_type = reader.read_byte("decimal.mantissa_type")?; @@ -110,7 +121,8 @@ fn decode_decimal<'a>(reader: &mut Reader<'a>, dicts: &WireDictionaries) -> Resu if bytes.len() > 1 { let second = bytes[1]; if (first == 0x00 && (second & 0x80) == 0) - || (first == 0xFF && (second & 0x80) != 0) { + || (first == 0xFF && (second & 0x80) != 0) + { return Err(DecodeError::DecimalMantissaNotMinimal); } } @@ -120,32 +132,14 @@ fn decode_decimal<'a>(reader: &mut Reader<'a>, dicts: &WireDictionaries) -> Resu } _ => { return Err(DecodeError::MalformedEncoding { - context: "invalid decimal mantissa type" + context: "invalid decimal mantissa type", }); } }; - // Validate normalization - match &mantissa { - DecimalMantissa::I64(v) => { - if *v == 0 { - if exponent != 0 { - return Err(DecodeError::DecimalNotNormalized); - } - } else if *v % 10 == 0 { - return Err(DecodeError::DecimalNotNormalized); - } - } - DecimalMantissa::Big(bytes) => { - if is_big_mantissa_zero(bytes) { - if exponent != 0 { - return Err(DecodeError::DecimalNotNormalized); - } - } else if is_big_mantissa_divisible_by_10(bytes) { - return Err(DecodeError::DecimalNotNormalized); - } - } - } + // Normalize on read: strip trailing zeros from mantissa and adjust exponent. + // This handles already-published edits with non-normalized decimals. + let (exponent, mantissa) = normalize_decimal(exponent, &mantissa); let unit_index = reader.read_varint("decimal.unit")? as usize; let unit = if unit_index == 0 { @@ -162,7 +156,11 @@ fn decode_decimal<'a>(reader: &mut Reader<'a>, dicts: &WireDictionaries) -> Resu Some(dicts.units[idx]) }; - Ok(Value::Decimal { exponent, mantissa, unit }) + Ok(Value::Decimal { + exponent, + mantissa, + unit, + }) } /// Checks if a big-endian two's complement mantissa represents zero. @@ -177,6 +175,7 @@ fn is_big_mantissa_zero(bytes: &[u8]) -> bool { /// Since 256 mod 10 = 6, we can compute iteratively: (carry * 6 + byte) mod 10. /// /// For negative numbers (high bit set), we need to handle two's complement. +#[cfg(test)] fn is_big_mantissa_divisible_by_10(bytes: &[u8]) -> bool { if bytes.is_empty() { return true; // Zero is divisible by 10 @@ -204,6 +203,7 @@ fn is_big_mantissa_divisible_by_10(bytes: &[u8]) -> bool { } /// Computes |x| mod 10 for a negative two's complement number. +#[cfg(test)] fn twos_complement_abs_mod_10(bytes: &[u8]) -> u32 { // Two's complement negation: invert all bits and add 1 // To get |x| mod 10, we compute (-x) mod 10 @@ -224,7 +224,10 @@ fn twos_complement_abs_mod_10(bytes: &[u8]) -> u32 { (remainder + 1) % 10 } -fn decode_text<'a>(reader: &mut Reader<'a>, dicts: &WireDictionaries) -> Result, DecodeError> { +fn decode_text<'a>( + reader: &mut Reader<'a>, + dicts: &WireDictionaries, +) -> Result, DecodeError> { let value = reader.read_str(MAX_STRING_LEN, "text")?; let lang_index = reader.read_varint("text.language")? as usize; @@ -242,7 +245,10 @@ fn decode_text<'a>(reader: &mut Reader<'a>, dicts: &WireDictionaries) -> Result< Some(dicts.languages[idx]) }; - Ok(Value::Text { value: Cow::Borrowed(value), language }) + Ok(Value::Text { + value: Cow::Borrowed(value), + language, + }) } fn decode_bytes<'a>(reader: &mut Reader<'a>) -> Result, DecodeError> { @@ -282,7 +288,7 @@ fn decode_time<'a>(reader: &mut Reader<'a>) -> Result, DecodeError> { // Read int48 as 6 bytes, sign-extend to i64 let time_micros_unsigned = u64::from_le_bytes([ - bytes[0], bytes[1], bytes[2], bytes[3], bytes[4], bytes[5], 0, 0 + bytes[0], bytes[1], bytes[2], bytes[3], bytes[4], bytes[5], 0, 0, ]); // Sign-extend from 48 bits let time_micros = if time_micros_unsigned & 0x8000_0000_0000 != 0 { @@ -316,8 +322,7 @@ fn decode_datetime<'a>(reader: &mut Reader<'a>) -> Result, DecodeError // DATETIME: 10 bytes (int64 epoch_micros + int16 offset_min), little-endian let bytes = reader.read_bytes(10, "datetime")?; let epoch_micros = i64::from_le_bytes([ - bytes[0], bytes[1], bytes[2], bytes[3], - bytes[4], bytes[5], bytes[6], bytes[7] + bytes[0], bytes[1], bytes[2], bytes[3], bytes[4], bytes[5], bytes[6], bytes[7], ]); let offset_min = i16::from_le_bytes([bytes[8], bytes[9]]); @@ -387,22 +392,41 @@ fn decode_rect<'a>(reader: &mut Reader<'a>) -> Result, DecodeError> { // Validate bounds if !(-90.0..=90.0).contains(&min_lat) || !(-90.0..=90.0).contains(&max_lat) { - return Err(DecodeError::LatitudeOutOfRange { lat: if !(-90.0..=90.0).contains(&min_lat) { min_lat } else { max_lat } }); + return Err(DecodeError::LatitudeOutOfRange { + lat: if !(-90.0..=90.0).contains(&min_lat) { + min_lat + } else { + max_lat + }, + }); } if !(-180.0..=180.0).contains(&min_lon) || !(-180.0..=180.0).contains(&max_lon) { - return Err(DecodeError::LongitudeOutOfRange { lon: if !(-180.0..=180.0).contains(&min_lon) { min_lon } else { max_lon } }); + return Err(DecodeError::LongitudeOutOfRange { + lon: if !(-180.0..=180.0).contains(&min_lon) { + min_lon + } else { + max_lon + }, + }); } if min_lat.is_nan() || min_lon.is_nan() || max_lat.is_nan() || max_lon.is_nan() { return Err(DecodeError::FloatIsNan); } - Ok(Value::Rect { min_lat, min_lon, max_lat, max_lon }) + Ok(Value::Rect { + min_lat, + min_lon, + max_lat, + max_lon, + }) } fn decode_embedding<'a>(reader: &mut Reader<'a>) -> Result, DecodeError> { let sub_type_byte = reader.read_byte("embedding.sub_type")?; - let sub_type = EmbeddingSubType::from_u8(sub_type_byte) - .ok_or(DecodeError::InvalidEmbeddingSubType { sub_type: sub_type_byte })?; + let sub_type = + EmbeddingSubType::from_u8(sub_type_byte).ok_or(DecodeError::InvalidEmbeddingSubType { + sub_type: sub_type_byte, + })?; let dims = reader.read_varint("embedding.dims")? as usize; if dims > MAX_EMBEDDING_DIMS { @@ -446,7 +470,11 @@ fn decode_embedding<'a>(reader: &mut Reader<'a>) -> Result, DecodeErro } } - Ok(Value::Embedding { sub_type, dims, data: Cow::Borrowed(data) }) + Ok(Value::Embedding { + sub_type, + dims, + data: Cow::Borrowed(data), + }) } /// Decodes a PropertyValue (property index + value + optional language). @@ -496,7 +524,11 @@ pub fn encode_value( let unit_index = dict_builder.add_unit(*unit); writer.write_varint(unit_index as u64); } - Value::Decimal { exponent, mantissa, unit } => { + Value::Decimal { + exponent, + mantissa, + unit, + } => { encode_decimal(writer, *exponent, mantissa)?; let unit_index = dict_builder.add_unit(*unit); writer.write_varint(unit_index as u64); @@ -511,18 +543,20 @@ pub fn encode_value( } Value::Date(s) => { // Parse RFC 3339 date string - let (days, offset_min) = parse_date_rfc3339(s).map_err(|_| EncodeError::InvalidInput { - context: "Invalid RFC 3339 date format", - })?; + let (days, offset_min) = + parse_date_rfc3339(s).map_err(|_| EncodeError::InvalidInput { + context: "Invalid RFC 3339 date format", + })?; // DATE: 6 bytes (int32 days + int16 offset_min), little-endian writer.write_bytes(&days.to_le_bytes()); writer.write_bytes(&offset_min.to_le_bytes()); } Value::Time(s) => { // Parse RFC 3339 time string - let (time_micros, offset_min) = parse_time_rfc3339(s).map_err(|_| EncodeError::InvalidInput { - context: "Invalid RFC 3339 time format", - })?; + let (time_micros, offset_min) = + parse_time_rfc3339(s).map_err(|_| EncodeError::InvalidInput { + context: "Invalid RFC 3339 time format", + })?; // Validate time_micros range (should already be validated by parser) if time_micros < 0 || time_micros > 86_399_999_999 { return Err(EncodeError::InvalidInput { @@ -537,9 +571,10 @@ pub fn encode_value( } Value::Datetime(s) => { // Parse RFC 3339 datetime string - let (epoch_micros, offset_min) = parse_datetime_rfc3339(s).map_err(|_| EncodeError::InvalidInput { - context: "Invalid RFC 3339 datetime format", - })?; + let (epoch_micros, offset_min) = + parse_datetime_rfc3339(s).map_err(|_| EncodeError::InvalidInput { + context: "Invalid RFC 3339 datetime format", + })?; // DATETIME: 10 bytes (int64 epoch_micros + int16 offset_min), little-endian writer.write_bytes(&epoch_micros.to_le_bytes()); writer.write_bytes(&offset_min.to_le_bytes()); @@ -573,12 +608,29 @@ pub fn encode_value( writer.write_f64(*a); } } - Value::Rect { min_lat, min_lon, max_lat, max_lon } => { + Value::Rect { + min_lat, + min_lon, + max_lat, + max_lon, + } => { if *min_lat < -90.0 || *min_lat > 90.0 || *max_lat < -90.0 || *max_lat > 90.0 { - return Err(EncodeError::LatitudeOutOfRange { lat: if *min_lat < -90.0 || *min_lat > 90.0 { *min_lat } else { *max_lat } }); + return Err(EncodeError::LatitudeOutOfRange { + lat: if *min_lat < -90.0 || *min_lat > 90.0 { + *min_lat + } else { + *max_lat + }, + }); } if *min_lon < -180.0 || *min_lon > 180.0 || *max_lon < -180.0 || *max_lon > 180.0 { - return Err(EncodeError::LongitudeOutOfRange { lon: if *min_lon < -180.0 || *min_lon > 180.0 { *min_lon } else { *max_lon } }); + return Err(EncodeError::LongitudeOutOfRange { + lon: if *min_lon < -180.0 || *min_lon > 180.0 { + *min_lon + } else { + *max_lon + }, + }); } if min_lat.is_nan() || min_lon.is_nan() || max_lat.is_nan() || max_lon.is_nan() { return Err(EncodeError::FloatIsNan); @@ -590,7 +642,11 @@ pub fn encode_value( writer.write_f64(*max_lat); writer.write_f64(*max_lon); } - Value::Embedding { sub_type, dims, data } => { + Value::Embedding { + sub_type, + dims, + data, + } => { let expected = sub_type.bytes_for_dims(*dims); if data.len() != expected { return Err(EncodeError::EmbeddingDimensionMismatch { @@ -616,36 +672,109 @@ pub fn encode_value( Ok(()) } -fn encode_decimal( - writer: &mut Writer, +/// Normalizes a decimal by stripping trailing zeros from the mantissa and +/// adjusting the exponent. Returns the normalized (exponent, mantissa) pair. +/// +/// Examples: +/// - (mantissa: 100, exponent: -2) → (mantissa: 1, exponent: 0) +/// - (mantissa: 1230, exponent: -2) → (mantissa: 123, exponent: -1) +/// - (mantissa: 0, exponent: 5) → (mantissa: 0, exponent: 0) +fn normalize_decimal( exponent: i32, mantissa: &DecimalMantissa<'_>, -) -> Result<(), EncodeError> { - // Validate normalization +) -> (i32, DecimalMantissa<'static>) { match mantissa { DecimalMantissa::I64(v) => { - if *v == 0 { - if exponent != 0 { - return Err(EncodeError::DecimalNotNormalized); - } - } else if *v % 10 == 0 { - return Err(EncodeError::DecimalNotNormalized); + let mut value = *v; + let mut exp = exponent; + + if value == 0 { + return (0, DecimalMantissa::I64(0)); + } + + while value != 0 && value % 10 == 0 { + value /= 10; + exp += 1; } + + (exp, DecimalMantissa::I64(value)) } DecimalMantissa::Big(bytes) => { if is_big_mantissa_zero(bytes) { - if exponent != 0 { - return Err(EncodeError::DecimalNotNormalized); - } - } else if is_big_mantissa_divisible_by_10(bytes) { - return Err(EncodeError::DecimalNotNormalized); + return (0, DecimalMantissa::I64(0)); + } + + // Convert big-endian two's complement to i128 for normalization. + // This handles values up to 128 bits which covers all practical cases. + let value = big_mantissa_to_i128(bytes); + let mut norm = value; + let mut exp = exponent; + + while norm != 0 && norm % 10 == 0 { + norm /= 10; + exp += 1; + } + + // If it fits in i64, use the more compact representation + if norm >= i64::MIN as i128 && norm <= i64::MAX as i128 { + (exp, DecimalMantissa::I64(norm as i64)) + } else { + ( + exp, + DecimalMantissa::Big(Cow::Owned(i128_to_minimal_twos_complement(norm))), + ) } } } +} + +/// Converts big-endian two's complement bytes to i128. +fn big_mantissa_to_i128(bytes: &[u8]) -> i128 { + if bytes.is_empty() { + return 0; + } + let is_negative = bytes[0] & 0x80 != 0; + let mut value: i128 = if is_negative { -1 } else { 0 }; + for &byte in bytes { + value = (value << 8) | byte as i128; + } + value +} + +/// Converts an i128 to minimal big-endian two's complement bytes. +fn i128_to_minimal_twos_complement(value: i128) -> Vec { + if value == 0 { + return vec![0]; + } - writer.write_signed_varint(exponent as i64); + let bytes = value.to_be_bytes(); - match mantissa { + // Find the first significant byte (skip redundant sign extension) + let mut start = 0; + if value > 0 { + while start < bytes.len() - 1 && bytes[start] == 0x00 && (bytes[start + 1] & 0x80) == 0 { + start += 1; + } + } else { + while start < bytes.len() - 1 && bytes[start] == 0xFF && (bytes[start + 1] & 0x80) != 0 { + start += 1; + } + } + + bytes[start..].to_vec() +} + +fn encode_decimal( + writer: &mut Writer, + exponent: i32, + mantissa: &DecimalMantissa<'_>, +) -> Result<(), EncodeError> { + // Normalize: strip trailing zeros from mantissa, adjust exponent + let (norm_exp, norm_mantissa) = normalize_decimal(exponent, mantissa); + + writer.write_signed_varint(norm_exp as i64); + + match &norm_mantissa { DecimalMantissa::I64(v) => { writer.write_byte(0x00); writer.write_signed_varint(*v); @@ -721,7 +850,10 @@ mod tests { #[test] fn test_integer_roundtrip() { for v in [0i64, 1, -1, i64::MAX, i64::MIN, 12345678] { - let value = Value::Integer { value: v, unit: None }; + let value = Value::Integer { + value: v, + unit: None, + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); @@ -738,7 +870,10 @@ mod tests { #[test] fn test_float_roundtrip() { for v in [0.0, 1.0, -1.0, f64::INFINITY, f64::NEG_INFINITY, 3.14159] { - let value = Value::Float { value: v, unit: None }; + let value = Value::Float { + value: v, + unit: None, + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); @@ -771,7 +906,16 @@ mod tests { // Compare inner values since one is Owned and one is Borrowed match (&value, &decoded) { - (Value::Text { value: v1, language: l1 }, Value::Text { value: v2, language: l2 }) => { + ( + Value::Text { + value: v1, + language: l1, + }, + Value::Text { + value: v2, + language: l2, + }, + ) => { assert_eq!(v1.as_ref(), v2.as_ref()); assert_eq!(l1, l2); } @@ -782,7 +926,11 @@ mod tests { #[test] fn test_point_roundtrip() { // 2D point (no altitude) - let value = Value::Point { lat: 37.7749, lon: -122.4194, alt: None }; + let value = Value::Point { + lat: 37.7749, + lon: -122.4194, + alt: None, + }; let dicts = WireDictionaries::default(); let mut dict_builder = DictionaryBuilder::new(); @@ -795,7 +943,11 @@ mod tests { assert_eq!(value, decoded); // 3D point (with altitude) - let value_3d = Value::Point { lat: 37.7749, lon: -122.4194, alt: Some(100.0) }; + let value_3d = Value::Point { + lat: 37.7749, + lon: -122.4194, + alt: Some(100.0), + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); @@ -810,21 +962,33 @@ mod tests { #[test] fn test_point_validation() { // Latitude out of range - let value = Value::Point { lat: 91.0, lon: 0.0, alt: None }; + let value = Value::Point { + lat: 91.0, + lon: 0.0, + alt: None, + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); let result = encode_value(&mut writer, &value, &mut dict_builder); assert!(result.is_err()); // Longitude out of range - let value = Value::Point { lat: 0.0, lon: 181.0, alt: None }; + let value = Value::Point { + lat: 0.0, + lon: 181.0, + alt: None, + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); let result = encode_value(&mut writer, &value, &mut dict_builder); assert!(result.is_err()); // NaN in altitude - let value = Value::Point { lat: 0.0, lon: 0.0, alt: Some(f64::NAN) }; + let value = Value::Point { + lat: 0.0, + lon: 0.0, + alt: Some(f64::NAN), + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); let result = encode_value(&mut writer, &value, &mut dict_builder); @@ -854,33 +1018,58 @@ mod tests { #[test] fn test_rect_validation() { // Latitude out of range - let value = Value::Rect { min_lat: -91.0, min_lon: 0.0, max_lat: 0.0, max_lon: 0.0 }; + let value = Value::Rect { + min_lat: -91.0, + min_lon: 0.0, + max_lat: 0.0, + max_lon: 0.0, + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); let result = encode_value(&mut writer, &value, &mut dict_builder); assert!(result.is_err()); - let value = Value::Rect { min_lat: 0.0, min_lon: 0.0, max_lat: 91.0, max_lon: 0.0 }; + let value = Value::Rect { + min_lat: 0.0, + min_lon: 0.0, + max_lat: 91.0, + max_lon: 0.0, + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); let result = encode_value(&mut writer, &value, &mut dict_builder); assert!(result.is_err()); // Longitude out of range - let value = Value::Rect { min_lat: 0.0, min_lon: -181.0, max_lat: 0.0, max_lon: 0.0 }; + let value = Value::Rect { + min_lat: 0.0, + min_lon: -181.0, + max_lat: 0.0, + max_lon: 0.0, + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); let result = encode_value(&mut writer, &value, &mut dict_builder); assert!(result.is_err()); - let value = Value::Rect { min_lat: 0.0, min_lon: 0.0, max_lat: 0.0, max_lon: 181.0 }; + let value = Value::Rect { + min_lat: 0.0, + min_lon: 0.0, + max_lat: 0.0, + max_lon: 181.0, + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); let result = encode_value(&mut writer, &value, &mut dict_builder); assert!(result.is_err()); // NaN not allowed - let value = Value::Rect { min_lat: f64::NAN, min_lon: 0.0, max_lat: 0.0, max_lon: 0.0 }; + let value = Value::Rect { + min_lat: f64::NAN, + min_lon: 0.0, + max_lat: 0.0, + max_lon: 0.0, + }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); let result = encode_value(&mut writer, &value, &mut dict_builder); @@ -893,7 +1082,10 @@ mod tests { let mut dict_builder = DictionaryBuilder::new(); // Simple iCalendar event (single occurrence) - let value = Value::Schedule(Cow::Owned("BEGIN:VEVENT\r\nDTSTART:20240315T090000Z\r\nDTEND:20240315T100000Z\r\nEND:VEVENT".to_string())); + let value = Value::Schedule(Cow::Owned( + "BEGIN:VEVENT\r\nDTSTART:20240315T090000Z\r\nDTEND:20240315T100000Z\r\nEND:VEVENT" + .to_string(), + )); let mut writer = Writer::new(); encode_value(&mut writer, &value, &mut dict_builder).unwrap(); @@ -928,8 +1120,16 @@ mod tests { // Compare inner values since one is Owned and one is Borrowed match (&value, &decoded) { ( - Value::Embedding { sub_type: s1, dims: d1, data: data1 }, - Value::Embedding { sub_type: s2, dims: d2, data: data2 }, + Value::Embedding { + sub_type: s1, + dims: d1, + data: data1, + }, + Value::Embedding { + sub_type: s2, + dims: d2, + data: data2, + }, ) => { assert_eq!(s1, s2); assert_eq!(d1, d2); @@ -941,7 +1141,9 @@ mod tests { #[test] fn test_decimal_normalized() { - // Valid: 12.34 = 1234 * 10^-2 + let dicts = WireDictionaries::default(); + + // Valid: 12.34 = 1234 * 10^-2 (already normalized) let valid = Value::Decimal { exponent: -2, mantissa: DecimalMantissa::I64(1234), @@ -950,16 +1152,117 @@ mod tests { let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); assert!(encode_value(&mut writer, &valid, &mut dict_builder).is_ok()); + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + match decoded { + Value::Decimal { + exponent, mantissa, .. + } => { + assert_eq!(exponent, -2); + assert_eq!(mantissa, DecimalMantissa::I64(1234)); + } + _ => panic!("expected Decimal"), + } - // Invalid: has trailing zeros - let invalid = Value::Decimal { + // Previously invalid (trailing zeros) — now normalized by encoder + let trailing = Value::Decimal { exponent: -2, mantissa: DecimalMantissa::I64(1230), unit: None, }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); - assert!(encode_value(&mut writer, &invalid, &mut dict_builder).is_err()); + assert!(encode_value(&mut writer, &trailing, &mut dict_builder).is_ok()); + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + match decoded { + Value::Decimal { + exponent, mantissa, .. + } => { + // 1230 * 10^-2 = 123 * 10^-1 + assert_eq!(exponent, -1); + assert_eq!(mantissa, DecimalMantissa::I64(123)); + } + _ => panic!("expected Decimal"), + } + } + + #[test] + fn test_decimal_normalize_zero() { + let dicts = WireDictionaries::default(); + + // Zero with non-zero exponent → normalized to (0, 0) + let zero = Value::Decimal { + exponent: 5, + mantissa: DecimalMantissa::I64(0), + unit: None, + }; + let mut dict_builder = DictionaryBuilder::new(); + let mut writer = Writer::new(); + assert!(encode_value(&mut writer, &zero, &mut dict_builder).is_ok()); + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + match decoded { + Value::Decimal { + exponent, mantissa, .. + } => { + assert_eq!(exponent, 0); + assert_eq!(mantissa, DecimalMantissa::I64(0)); + } + _ => panic!("expected Decimal"), + } + } + + #[test] + fn test_decimal_normalize_100_exp_neg2() { + let dicts = WireDictionaries::default(); + + // $1.00 = 100 * 10^-2 → normalized to 1 * 10^0 + let dollar = Value::Decimal { + exponent: -2, + mantissa: DecimalMantissa::I64(100), + unit: None, + }; + let mut dict_builder = DictionaryBuilder::new(); + let mut writer = Writer::new(); + assert!(encode_value(&mut writer, &dollar, &mut dict_builder).is_ok()); + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + match decoded { + Value::Decimal { + exponent, mantissa, .. + } => { + assert_eq!(exponent, 0); + assert_eq!(mantissa, DecimalMantissa::I64(1)); + } + _ => panic!("expected Decimal"), + } + } + + #[test] + fn test_decimal_normalize_negative() { + let dicts = WireDictionaries::default(); + + // -500 * 10^0 → -5 * 10^2 + let neg = Value::Decimal { + exponent: 0, + mantissa: DecimalMantissa::I64(-500), + unit: None, + }; + let mut dict_builder = DictionaryBuilder::new(); + let mut writer = Writer::new(); + assert!(encode_value(&mut writer, &neg, &mut dict_builder).is_ok()); + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + match decoded { + Value::Decimal { + exponent, mantissa, .. + } => { + assert_eq!(exponent, 2); + assert_eq!(mantissa, DecimalMantissa::I64(-5)); + } + _ => panic!("expected Decimal"), + } } #[test] @@ -969,10 +1272,10 @@ mod tests { // Test various date values (RFC 3339 format) let test_cases = [ - "1970-01-01Z", // Unix epoch, UTC - "2024-03-15Z", // March 15, 2024 UTC - "2024-03-15+05:30", // March 15, 2024 +05:30 - "2024-03-15-08:00", // March 15, 2024 -08:00 + "1970-01-01Z", // Unix epoch, UTC + "2024-03-15Z", // March 15, 2024 UTC + "2024-03-15+05:30", // March 15, 2024 +05:30 + "2024-03-15-08:00", // March 15, 2024 -08:00 ]; for date_str in test_cases { @@ -987,7 +1290,12 @@ mod tests { // Compare the string values match (&value, &decoded) { (Value::Date(v1), Value::Date(v2)) => { - assert_eq!(v1.as_ref(), v2.as_ref(), "Roundtrip failed for {}", date_str); + assert_eq!( + v1.as_ref(), + v2.as_ref(), + "Roundtrip failed for {}", + date_str + ); } _ => panic!("expected Date values"), } @@ -1001,11 +1309,11 @@ mod tests { // Test various time values (RFC 3339 format) let test_cases = [ - "00:00:00Z", // Midnight UTC - "14:30:00Z", // 14:30:00 UTC - "14:30:00.5+05:30", // 14:30:00.500 +05:30 - "23:59:59.999999Z", // 23:59:59.999999 UTC - "00:00:00-05:00", // Midnight -05:00 + "00:00:00Z", // Midnight UTC + "14:30:00Z", // 14:30:00 UTC + "14:30:00.5+05:30", // 14:30:00.500 +05:30 + "23:59:59.999999Z", // 23:59:59.999999 UTC + "00:00:00-05:00", // Midnight -05:00 ]; for time_str in test_cases { @@ -1020,7 +1328,12 @@ mod tests { // Compare the string values match (&value, &decoded) { (Value::Time(v1), Value::Time(v2)) => { - assert_eq!(v1.as_ref(), v2.as_ref(), "Roundtrip failed for {}", time_str); + assert_eq!( + v1.as_ref(), + v2.as_ref(), + "Roundtrip failed for {}", + time_str + ); } _ => panic!("expected Time values"), } @@ -1034,10 +1347,10 @@ mod tests { // Test various datetime values (RFC 3339 format) let test_cases = [ - "1970-01-01T00:00:00Z", // Unix epoch UTC - "2024-03-15T14:30:00Z", // 2024-03-15T14:30:00Z - "2024-03-15T14:30:00+05:30", // 2024-03-15T14:30:00+05:30 - "2024-03-15T14:30:00.123456Z", // With microseconds + "1970-01-01T00:00:00Z", // Unix epoch UTC + "2024-03-15T14:30:00Z", // 2024-03-15T14:30:00Z + "2024-03-15T14:30:00+05:30", // 2024-03-15T14:30:00+05:30 + "2024-03-15T14:30:00.123456Z", // With microseconds ]; for datetime_str in test_cases { @@ -1052,7 +1365,12 @@ mod tests { // Compare the string values match (&value, &decoded) { (Value::Datetime(v1), Value::Datetime(v2)) => { - assert_eq!(v1.as_ref(), v2.as_ref(), "Roundtrip failed for {}", datetime_str); + assert_eq!( + v1.as_ref(), + v2.as_ref(), + "Roundtrip failed for {}", + datetime_str + ); } _ => panic!("expected Datetime values"), } @@ -1138,17 +1456,19 @@ mod tests { // Test negative numbers (two's complement) // -10 in two's complement (1 byte): 0xF6 assert!(is_big_mantissa_divisible_by_10(&[0xF6])); // -10 - // -20 in two's complement (1 byte): 0xEC + // -20 in two's complement (1 byte): 0xEC assert!(is_big_mantissa_divisible_by_10(&[0xEC])); // -20 - // -1 in two's complement (1 byte): 0xFF + // -1 in two's complement (1 byte): 0xFF assert!(!is_big_mantissa_divisible_by_10(&[0xFF])); // -1 - // -7 in two's complement (1 byte): 0xF9 + // -7 in two's complement (1 byte): 0xF9 assert!(!is_big_mantissa_divisible_by_10(&[0xF9])); // -7 } #[test] fn test_big_decimal_normalization_encode() { - // Valid: mantissa not divisible by 10 + let dicts = WireDictionaries::default(); + + // Valid: mantissa not divisible by 10 (already normalized) let valid = Value::Decimal { exponent: 0, mantissa: DecimalMantissa::Big(Cow::Owned(vec![0x07])), // 7 @@ -1157,28 +1477,62 @@ mod tests { let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); assert!(encode_value(&mut writer, &valid, &mut dict_builder).is_ok()); + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + match decoded { + Value::Decimal { + exponent, mantissa, .. + } => { + assert_eq!(exponent, 0); + assert_eq!(mantissa, DecimalMantissa::I64(7)); // Big(7) normalized to I64(7) + } + _ => panic!("expected Decimal"), + } - // Invalid: mantissa is 10 (divisible by 10) - let invalid = Value::Decimal { + // Previously invalid: mantissa 10 (divisible by 10) — now normalized + let was_invalid = Value::Decimal { exponent: 0, mantissa: DecimalMantissa::Big(Cow::Owned(vec![0x0A])), // 10 unit: None, }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); - assert!(encode_value(&mut writer, &invalid, &mut dict_builder).is_err()); + assert!(encode_value(&mut writer, &was_invalid, &mut dict_builder).is_ok()); + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + match decoded { + Value::Decimal { + exponent, mantissa, .. + } => { + // 10 * 10^0 = 1 * 10^1 + assert_eq!(exponent, 1); + assert_eq!(mantissa, DecimalMantissa::I64(1)); + } + _ => panic!("expected Decimal"), + } - // Invalid: zero mantissa with non-zero exponent - let invalid_zero = Value::Decimal { + // Previously invalid: zero mantissa with non-zero exponent — now normalized + let was_invalid_zero = Value::Decimal { exponent: 1, mantissa: DecimalMantissa::Big(Cow::Owned(vec![0x00])), unit: None, }; let mut dict_builder = DictionaryBuilder::new(); let mut writer = Writer::new(); - assert!(encode_value(&mut writer, &invalid_zero, &mut dict_builder).is_err()); + assert!(encode_value(&mut writer, &was_invalid_zero, &mut dict_builder).is_ok()); + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + match decoded { + Value::Decimal { + exponent, mantissa, .. + } => { + assert_eq!(exponent, 0); + assert_eq!(mantissa, DecimalMantissa::I64(0)); + } + _ => panic!("expected Decimal"), + } - // Valid: zero mantissa with zero exponent + // Valid: zero mantissa with zero exponent (already normalized) let valid_zero = Value::Decimal { exponent: 0, mantissa: DecimalMantissa::Big(Cow::Owned(vec![0x00])), @@ -1188,5 +1542,4 @@ mod tests { let mut writer = Writer::new(); assert!(encode_value(&mut writer, &valid_zero, &mut dict_builder).is_ok()); } - } diff --git a/typescript/src/codec/value.ts b/typescript/src/codec/value.ts index 8d7dbf5..aaab4f0 100644 --- a/typescript/src/codec/value.ts +++ b/typescript/src/codec/value.ts @@ -150,17 +150,119 @@ export function encodeValuePayload(writer: Writer, value: Value): void { } /** - * Encodes a decimal value. + * Normalizes a decimal value by stripping trailing zeros from the mantissa + * and adjusting the exponent accordingly. + * + * Examples: + * - (mantissa: 100, exponent: -2) → (mantissa: 1, exponent: 0) + * - (mantissa: 1230, exponent: -2) → (mantissa: 123, exponent: -1) + * - (mantissa: 0, exponent: 5) → (mantissa: 0, exponent: 0) + */ +function normalizeDecimal( + exponent: number, + mantissa: DecimalMantissa +): { exponent: number; mantissa: DecimalMantissa } { + if (mantissa.type === "i64") { + let value = mantissa.value; + let exp = exponent; + + // Zero mantissa must have exponent 0 + if (value === 0n) { + return { exponent: 0, mantissa: { type: "i64", value: 0n } }; + } + + // Strip trailing zeros + while (value !== 0n && value % 10n === 0n) { + value = value / 10n; + exp += 1; + } + + return { exponent: exp, mantissa: { type: "i64", value } }; + } else { + // Big mantissa (big-endian two's complement bytes) + const bytes = mantissa.bytes; + + // Check if zero + if (bytes.every((b) => b === 0)) { + return { exponent: 0, mantissa: { type: "i64", value: 0n } }; + } + + // Convert big-endian two's complement to BigInt for normalization + const isNegative = bytes.length > 0 && (bytes[0] & 0x80) !== 0; + let value = 0n; + for (const byte of bytes) { + value = (value << 8n) | BigInt(byte); + } + // If negative, sign-extend: the bytes represent a two's complement value + if (isNegative) { + value = value - (1n << BigInt(bytes.length * 8)); + } + + // Strip trailing zeros + let exp = exponent; + while (value !== 0n && value % 10n === 0n) { + value = value / 10n; + exp += 1; + } + + // If it fits in i64 range, use i64 encoding (more compact) + if (value >= -9223372036854775808n && value <= 9223372036854775807n) { + return { exponent: exp, mantissa: { type: "i64", value } }; + } + + // Convert back to big-endian two's complement bytes (minimal encoding) + const resultBytes = bigIntToMinimalTwosComplement(value); + return { exponent: exp, mantissa: { type: "big", bytes: resultBytes } }; + } +} + +/** + * Converts a BigInt to minimal big-endian two's complement bytes. + */ +function bigIntToMinimalTwosComplement(value: bigint): Uint8Array { + if (value === 0n) { + return new Uint8Array([0]); + } + + const isNegative = value < 0n; + // Work with the absolute value to determine byte count, + // then encode in two's complement + const abs = isNegative ? -value : value; + + // Determine how many bytes we need + const bitLen = abs.toString(2).length; + // +1 for sign bit, then round up to bytes + const byteLen = Math.ceil((bitLen + 1) / 8); + + // Encode in two's complement + let encoded = isNegative + ? (1n << BigInt(byteLen * 8)) + value // two's complement for negative + : value; + + const bytes = new Uint8Array(byteLen); + for (let i = byteLen - 1; i >= 0; i--) { + bytes[i] = Number(encoded & 0xFFn); + encoded >>= 8n; + } + + return bytes; +} + +/** + * Encodes a decimal value. Normalizes the mantissa/exponent before encoding + * to ensure trailing zeros are stripped. */ function encodeDecimal(writer: Writer, exponent: number, mantissa: DecimalMantissa): void { - writer.writeSignedVarint(BigInt(exponent)); + const normalized = normalizeDecimal(exponent, mantissa); - if (mantissa.type === "i64") { + writer.writeSignedVarint(BigInt(normalized.exponent)); + + if (normalized.mantissa.type === "i64") { writer.writeByte(0x00); // mantissa_type = varint - writer.writeSignedVarint(mantissa.value); + writer.writeSignedVarint(normalized.mantissa.value); } else { writer.writeByte(0x01); // mantissa_type = bytes - writer.writeLengthPrefixedBytes(mantissa.bytes); + writer.writeLengthPrefixedBytes(normalized.mantissa.bytes); } } diff --git a/typescript/src/test/basic.test.ts b/typescript/src/test/basic.test.ts index 5aa364f..f52f0b3 100644 --- a/typescript/src/test/basic.test.ts +++ b/typescript/src/test/basic.test.ts @@ -1257,6 +1257,222 @@ describe("Codec", () => { expect(() => encodeEdit(edit)).toThrow("Invalid RFC 3339 time"); }); +describe("Decimal normalization", () => { + it("normalizes i64 mantissa with trailing zeros", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + // mantissa 100, exponent -2 represents 1.00 — should normalize to mantissa 1, exponent 0 + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: -2, mantissa: { type: "i64", value: 100n } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + expect(op.type).toBe("createEntity"); + if (op.type === "createEntity") { + const val = op.values[0].value; + expect(val.type).toBe("decimal"); + if (val.type === "decimal") { + // Should be normalized: 100 * 10^-2 = 1 * 10^0 + expect(val.mantissa).toEqual({ type: "i64", value: 1n }); + expect(val.exponent).toBe(0); + } + } + }); + + it("normalizes i64 mantissa 1230 with exponent -2", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: -2, mantissa: { type: "i64", value: 1230n } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + if (op.type === "createEntity") { + const val = op.values[0].value; + if (val.type === "decimal") { + // 1230 * 10^-2 = 123 * 10^-1 + expect(val.mantissa).toEqual({ type: "i64", value: 123n }); + expect(val.exponent).toBe(-1); + } + } + }); + + it("normalizes zero mantissa with non-zero exponent", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: 5, mantissa: { type: "i64", value: 0n } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + if (op.type === "createEntity") { + const val = op.values[0].value; + if (val.type === "decimal") { + expect(val.mantissa).toEqual({ type: "i64", value: 0n }); + expect(val.exponent).toBe(0); + } + } + }); + + it("does not modify already-normalized decimals", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: -2, mantissa: { type: "i64", value: 12345n } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + if (op.type === "createEntity") { + const val = op.values[0].value; + if (val.type === "decimal") { + expect(val.mantissa).toEqual({ type: "i64", value: 12345n }); + expect(val.exponent).toBe(-2); + } + } + }); + + it("normalizes negative mantissa with trailing zeros", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: 0, mantissa: { type: "i64", value: -500n } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + if (op.type === "createEntity") { + const val = op.values[0].value; + if (val.type === "decimal") { + // -500 * 10^0 = -5 * 10^2 + expect(val.mantissa).toEqual({ type: "i64", value: -5n }); + expect(val.exponent).toBe(2); + } + } + }); + + it("normalizes big mantissa with trailing zeros", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + // Big-endian two's complement for 500: 0x01F4 + const bigBytes = new Uint8Array([0x01, 0xF4]); + + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: 0, mantissa: { type: "big", bytes: bigBytes } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + if (op.type === "createEntity") { + const val = op.values[0].value; + if (val.type === "decimal") { + // 500 * 10^0 = 5 * 10^2, and 5 fits in i64 + expect(val.mantissa).toEqual({ type: "i64", value: 5n }); + expect(val.exponent).toBe(2); + } + } + }); +}); + describe("Compression", () => { it("isCompressed detects GRC2Z magic", () => { const compressed = new Uint8Array([0x47, 0x52, 0x43, 0x32, 0x5a, 0x00]); // "GRC2Z" + data From 76357f9367705bc7edd598cae568b41fbbbfd362 Mon Sep 17 00:00:00 2001 From: Nik Graf Date: Tue, 10 Mar 2026 08:17:47 +0100 Subject: [PATCH 2/9] Align decimal normalization across Rust and TypeScript MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Normalize DECIMAL values consistently on encode and decode in both runtimes. Replace Rust’s lossy i128-based big mantissa normalization with arbitrary- precision byte-level normalization, preserve borrowed canonical mantissas on decode, and fix TypeScript’s two’s-complement minimization for negative power-of-two boundary values. Add regression tests for large big mantissas, decode-time normalization, and minimal negative big-byte encodings. Update the spec and encoding docs to reflect canonical encoding plus lenient decode-time normalization for legacy payloads. --- encoding.md | 5 +- rust/crates/grc-20/src/codec/value.rs | 333 +++++++++++++++++++++----- spec.md | 2 +- typescript/src/codec/value.ts | 77 ++++-- typescript/src/test/basic.test.ts | 55 +++++ 5 files changed, 392 insertions(+), 80 deletions(-) diff --git a/encoding.md b/encoding.md index 82df368..597497b 100644 --- a/encoding.md +++ b/encoding.md @@ -316,7 +316,9 @@ Embedding: > - If mantissa fits in signed 64-bit integer (-2^63 to 2^63-1), `mantissa_type` MUST be `0x00` (varint). > - `mantissa_type = 0x01` (bytes) is reserved for values outside int64 range. > - When `mantissa_type = 0x01`, mantissa bytes MUST be big-endian two's complement, minimal-length (no redundant sign extension). -> - Non-compliant encodings MUST be rejected (E005). +> - Encoders MUST emit normalized DECIMAL values. +> - Decoders SHOULD accept legacy non-normalized DECIMAL values and normalize them after decoding. +> - Non-minimal mantissa bytes MUST be rejected (E005). --- @@ -417,7 +419,6 @@ Indexers MUST reject edits that fail structural validation: | Varint encoding | Overlong encoding or exceeds 10 bytes | | Reserved bits | Non-zero | | Mantissa bytes | Non-minimal encoding | -| DECIMAL normalization | Mantissa has trailing zeros, or zero not encoded as {0,0} | | Signatures | Invalid (if governance requires) | | BOOLEAN values | Not 0x00 or 0x01 | | POINT bounds | Latitude outside [-90, +90] or longitude outside [-180, +180] | diff --git a/rust/crates/grc-20/src/codec/value.rs b/rust/crates/grc-20/src/codec/value.rs index bb5176d..23ce185 100644 --- a/rust/crates/grc-20/src/codec/value.rs +++ b/rust/crates/grc-20/src/codec/value.rs @@ -137,9 +137,9 @@ fn decode_decimal<'a>( } }; - // Normalize on read: strip trailing zeros from mantissa and adjust exponent. - // This handles already-published edits with non-normalized decimals. - let (exponent, mantissa) = normalize_decimal(exponent, &mantissa); + // Normalize on read to handle already-published edits with non-normalized + // decimals, while preserving borrowed bytes for canonical big mantissas. + let (exponent, mantissa) = normalize_decimal_for_decode(exponent, mantissa); let unit_index = reader.read_varint("decimal.unit")? as usize; let unit = if unit_index == 0 { @@ -175,7 +175,6 @@ fn is_big_mantissa_zero(bytes: &[u8]) -> bool { /// Since 256 mod 10 = 6, we can compute iteratively: (carry * 6 + byte) mod 10. /// /// For negative numbers (high bit set), we need to handle two's complement. -#[cfg(test)] fn is_big_mantissa_divisible_by_10(bytes: &[u8]) -> bool { if bytes.is_empty() { return true; // Zero is divisible by 10 @@ -203,7 +202,6 @@ fn is_big_mantissa_divisible_by_10(bytes: &[u8]) -> bool { } /// Computes |x| mod 10 for a negative two's complement number. -#[cfg(test)] fn twos_complement_abs_mod_10(bytes: &[u8]) -> u32 { // Two's complement negation: invert all bits and add 1 // To get |x| mod 10, we compute (-x) mod 10 @@ -672,14 +670,27 @@ pub fn encode_value( Ok(()) } -/// Normalizes a decimal by stripping trailing zeros from the mantissa and -/// adjusting the exponent. Returns the normalized (exponent, mantissa) pair. +/// Returns whether the mantissa must be canonicalized to satisfy the +/// normalization rules or the preferred wire representation. /// -/// Examples: -/// - (mantissa: 100, exponent: -2) → (mantissa: 1, exponent: 0) -/// - (mantissa: 1230, exponent: -2) → (mantissa: 123, exponent: -1) -/// - (mantissa: 0, exponent: 5) → (mantissa: 0, exponent: 0) -fn normalize_decimal( +/// This is true when: +/// - zero is not encoded as `I64(0)` with exponent 0 +/// - the mantissa still has trailing decimal zeros +/// - a big mantissa now fits in `i64` and should use the compact varint form +fn decimal_needs_canonicalization(exponent: i32, mantissa: &DecimalMantissa<'_>) -> bool { + match mantissa { + DecimalMantissa::I64(v) => (*v == 0 && exponent != 0) || (*v != 0 && *v % 10 == 0), + DecimalMantissa::Big(bytes) => { + is_big_mantissa_zero(bytes) + || is_big_mantissa_divisible_by_10(bytes) + || twos_complement_bytes_to_i64(bytes).is_some() + } + } +} + +/// Normalizes a decimal by stripping trailing zeros from the mantissa and +/// adjusting the exponent. Returns an owned canonical representation. +fn normalize_decimal_owned( exponent: i32, mantissa: &DecimalMantissa<'_>, ) -> (i32, DecimalMantissa<'static>) { @@ -704,64 +715,151 @@ fn normalize_decimal( return (0, DecimalMantissa::I64(0)); } - // Convert big-endian two's complement to i128 for normalization. - // This handles values up to 128 bits which covers all practical cases. - let value = big_mantissa_to_i128(bytes); - let mut norm = value; + let is_negative = bytes[0] & 0x80 != 0; + let mut magnitude = twos_complement_abs_bytes(bytes); let mut exp = exponent; - while norm != 0 && norm % 10 == 0 { - norm /= 10; + loop { + let (quotient, remainder) = divide_unsigned_be_by_10(&magnitude); + if remainder != 0 { + break; + } + magnitude = quotient; exp += 1; } - // If it fits in i64, use the more compact representation - if norm >= i64::MIN as i128 && norm <= i64::MAX as i128 { - (exp, DecimalMantissa::I64(norm as i64)) + let canonical_bytes = signed_bytes_from_magnitude(is_negative, &magnitude); + if let Some(value) = twos_complement_bytes_to_i64(&canonical_bytes) { + (exp, DecimalMantissa::I64(value)) } else { - ( - exp, - DecimalMantissa::Big(Cow::Owned(i128_to_minimal_twos_complement(norm))), - ) + (exp, DecimalMantissa::Big(Cow::Owned(canonical_bytes))) } } } } -/// Converts big-endian two's complement bytes to i128. -fn big_mantissa_to_i128(bytes: &[u8]) -> i128 { +fn normalize_decimal_for_decode<'a>( + exponent: i32, + mantissa: DecimalMantissa<'a>, +) -> (i32, DecimalMantissa<'a>) { + if !decimal_needs_canonicalization(exponent, &mantissa) { + return (exponent, mantissa); + } + + let (exp, canonical) = normalize_decimal_owned(exponent, &mantissa); + match canonical { + DecimalMantissa::I64(value) => (exp, DecimalMantissa::I64(value)), + DecimalMantissa::Big(bytes) => (exp, DecimalMantissa::Big(Cow::Owned(bytes.into_owned()))), + } +} + +fn twos_complement_abs_bytes(bytes: &[u8]) -> Vec { if bytes.is_empty() { - return 0; + return Vec::new(); } - let is_negative = bytes[0] & 0x80 != 0; - let mut value: i128 = if is_negative { -1 } else { 0 }; + + if bytes[0] & 0x80 == 0 { + let first_non_zero = bytes + .iter() + .position(|&byte| byte != 0) + .unwrap_or(bytes.len()); + return bytes[first_non_zero..].to_vec(); + } + + let mut magnitude = vec![0u8; bytes.len()]; + let mut carry = 1u16; + for (index, &byte) in bytes.iter().enumerate().rev() { + let sum = (!byte as u16) + carry; + magnitude[index] = sum as u8; + carry = sum >> 8; + } + + let first_non_zero = magnitude + .iter() + .position(|&byte| byte != 0) + .unwrap_or(magnitude.len()); + magnitude[first_non_zero..].to_vec() +} + +fn divide_unsigned_be_by_10(bytes: &[u8]) -> (Vec, u8) { + let mut quotient = Vec::with_capacity(bytes.len()); + let mut remainder = 0u16; + for &byte in bytes { - value = (value << 8) | byte as i128; + let current = (remainder << 8) | byte as u16; + quotient.push((current / 10) as u8); + remainder = current % 10; } - value + + let first_non_zero = quotient + .iter() + .position(|&byte| byte != 0) + .unwrap_or(quotient.len()); + (quotient[first_non_zero..].to_vec(), remainder as u8) } -/// Converts an i128 to minimal big-endian two's complement bytes. -fn i128_to_minimal_twos_complement(value: i128) -> Vec { - if value == 0 { +fn signed_bytes_from_magnitude(is_negative: bool, magnitude: &[u8]) -> Vec { + if magnitude.is_empty() { return vec![0]; } - let bytes = value.to_be_bytes(); + let first_non_zero = magnitude + .iter() + .position(|&byte| byte != 0) + .unwrap_or(magnitude.len()); + let magnitude = &magnitude[first_non_zero..]; + if magnitude.is_empty() { + return vec![0]; + } - // Find the first significant byte (skip redundant sign extension) - let mut start = 0; - if value > 0 { - while start < bytes.len() - 1 && bytes[start] == 0x00 && (bytes[start + 1] & 0x80) == 0 { - start += 1; + if !is_negative { + let mut bytes = magnitude.to_vec(); + if bytes[0] & 0x80 != 0 { + bytes.insert(0, 0x00); } + return bytes; + } + + let fits_in_current_width = magnitude[0] < 0x80 + || (magnitude[0] == 0x80 && magnitude[1..].iter().all(|&byte| byte == 0)); + let width = if fits_in_current_width { + magnitude.len() } else { - while start < bytes.len() - 1 && bytes[start] == 0xFF && (bytes[start + 1] & 0x80) != 0 { - start += 1; - } + magnitude.len() + 1 + }; + + let mut bytes = vec![0u8; width]; + bytes[width - magnitude.len()..].copy_from_slice(magnitude); + for byte in &mut bytes { + *byte = !*byte; + } + + let mut carry = 1u16; + for byte in bytes.iter_mut().rev() { + let sum = *byte as u16 + carry; + *byte = sum as u8; + carry = sum >> 8; + } + + while bytes.len() > 1 && bytes[0] == 0xFF && (bytes[1] & 0x80) != 0 { + bytes.remove(0); + } + + bytes +} + +fn twos_complement_bytes_to_i64(bytes: &[u8]) -> Option { + if bytes.is_empty() { + return Some(0); + } + if bytes.len() > 8 { + return None; } - bytes[start..].to_vec() + let fill = if bytes[0] & 0x80 != 0 { 0xFF } else { 0x00 }; + let mut expanded = [fill; 8]; + expanded[8 - bytes.len()..].copy_from_slice(bytes); + Some(i64::from_be_bytes(expanded)) } fn encode_decimal( @@ -769,20 +867,36 @@ fn encode_decimal( exponent: i32, mantissa: &DecimalMantissa<'_>, ) -> Result<(), EncodeError> { - // Normalize: strip trailing zeros from mantissa, adjust exponent - let (norm_exp, norm_mantissa) = normalize_decimal(exponent, mantissa); + // Normalize: strip trailing zeros from mantissa, adjust exponent, and use + // the compact i64 wire form whenever the canonical mantissa fits in i64. + if !decimal_needs_canonicalization(exponent, mantissa) { + writer.write_signed_varint(exponent as i64); + match mantissa { + DecimalMantissa::I64(value) => { + writer.write_byte(0x00); + writer.write_signed_varint(*value); + } + DecimalMantissa::Big(bytes) => { + writer.write_byte(0x01); + writer.write_varint(bytes.len() as u64); + writer.write_bytes(bytes); + } + } + return Ok(()); + } + let (norm_exp, norm_mantissa) = normalize_decimal_owned(exponent, mantissa); writer.write_signed_varint(norm_exp as i64); - match &norm_mantissa { - DecimalMantissa::I64(v) => { + match norm_mantissa { + DecimalMantissa::I64(value) => { writer.write_byte(0x00); - writer.write_signed_varint(*v); + writer.write_signed_varint(value); } DecimalMantissa::Big(bytes) => { writer.write_byte(0x01); writer.write_varint(bytes.len() as u64); - writer.write_bytes(bytes); + writer.write_bytes(bytes.as_ref()); } } @@ -1456,11 +1570,11 @@ mod tests { // Test negative numbers (two's complement) // -10 in two's complement (1 byte): 0xF6 assert!(is_big_mantissa_divisible_by_10(&[0xF6])); // -10 - // -20 in two's complement (1 byte): 0xEC + // -20 in two's complement (1 byte): 0xEC assert!(is_big_mantissa_divisible_by_10(&[0xEC])); // -20 - // -1 in two's complement (1 byte): 0xFF + // -1 in two's complement (1 byte): 0xFF assert!(!is_big_mantissa_divisible_by_10(&[0xFF])); // -1 - // -7 in two's complement (1 byte): 0xF9 + // -7 in two's complement (1 byte): 0xF9 assert!(!is_big_mantissa_divisible_by_10(&[0xF9])); // -7 } @@ -1542,4 +1656,115 @@ mod tests { let mut writer = Writer::new(); assert!(encode_value(&mut writer, &valid_zero, &mut dict_builder).is_ok()); } + + #[test] + fn test_decode_decimal_normalizes_large_big_mantissa_without_truncation() { + let dicts = WireDictionaries::default(); + let large = vec![ + 0x0A, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, + ]; + + let mut writer = Writer::new(); + writer.write_signed_varint(0); + writer.write_byte(0x01); + writer.write_varint(large.len() as u64); + writer.write_bytes(&large); + writer.write_varint(0); + + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + + match decoded { + Value::Decimal { + exponent, + mantissa: DecimalMantissa::Big(bytes), + unit, + } => { + assert_eq!(exponent, 1); + assert_eq!( + bytes.as_ref(), + &[ + 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00 + ] + ); + assert_eq!(unit, None); + } + other => panic!("expected large normalized Decimal, got {other:?}"), + } + } + + #[test] + fn test_decode_decimal_keeps_borrowed_large_big_mantissa_when_already_canonical() { + let dicts = WireDictionaries::default(); + let canonical = vec![ + 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, + ]; + + let mut writer = Writer::new(); + writer.write_signed_varint(0); + writer.write_byte(0x01); + writer.write_varint(canonical.len() as u64); + writer.write_bytes(&canonical); + writer.write_varint(0); + + let encoded = writer.as_bytes().to_vec(); + let mut reader = Reader::new(&encoded); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + + match decoded { + Value::Decimal { + exponent, + mantissa: DecimalMantissa::Big(bytes), + unit, + } => { + assert_eq!(exponent, 0); + assert_eq!(bytes.as_ref(), canonical.as_slice()); + assert!(matches!(bytes, Cow::Borrowed(_))); + assert_eq!(unit, None); + } + other => panic!("expected borrowed large Decimal, got {other:?}"), + } + } + + #[test] + fn test_encode_decimal_normalizes_large_big_mantissa_without_truncation() { + let dicts = WireDictionaries::default(); + let mut dict_builder = DictionaryBuilder::new(); + let large = vec![ + 0x0A, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, + ]; + let value = Value::Decimal { + exponent: 0, + mantissa: DecimalMantissa::Big(Cow::Owned(large)), + unit: None, + }; + + let mut writer = Writer::new(); + encode_value(&mut writer, &value, &mut dict_builder).unwrap(); + + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + + match decoded { + Value::Decimal { + exponent, + mantissa: DecimalMantissa::Big(bytes), + .. + } => { + assert_eq!(exponent, 1); + assert_eq!( + bytes.as_ref(), + &[ + 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00 + ] + ); + } + other => panic!("expected large normalized Decimal, got {other:?}"), + } + } } diff --git a/spec.md b/spec.md index 4d5fdd7..c541ea1 100644 --- a/spec.md +++ b/spec.md @@ -186,7 +186,7 @@ Examples: - `$12.34` → `{ exponent: -2, mantissa: 1234 }` - `0.000001` → `{ exponent: -6, mantissa: 1 }` -> **NORMATIVE:** DECIMAL values MUST be encoded in normalized form: mantissa has no trailing zeros, and zero is represented as `{ exponent: 0, mantissa: 0 }`. This ensures deterministic encoding for content addressing. +> **NORMATIVE:** DECIMAL values MUST be encoded in normalized form: mantissa has no trailing zeros, and zero is represented as `{ exponent: 0, mantissa: 0 }`. This ensures deterministic encoding for content addressing. Decoders and indexers SHOULD accept legacy non-normalized DECIMAL payloads and normalize them after decoding so already-published data remains interoperable. Applications needing to preserve display precision (e.g., "12.30" vs "12.3") can use the property's format to indicate the intended presentation. The format is specified on the property entity in the knowledge graph. The normalized DECIMAL ensures deterministic encoding; the property's format controls how all its values render. diff --git a/typescript/src/codec/value.ts b/typescript/src/codec/value.ts index aaab4f0..b4c21f9 100644 --- a/typescript/src/codec/value.ts +++ b/typescript/src/codec/value.ts @@ -179,39 +179,30 @@ function normalizeDecimal( return { exponent: exp, mantissa: { type: "i64", value } }; } else { - // Big mantissa (big-endian two's complement bytes) const bytes = mantissa.bytes; - // Check if zero if (bytes.every((b) => b === 0)) { return { exponent: 0, mantissa: { type: "i64", value: 0n } }; } - // Convert big-endian two's complement to BigInt for normalization - const isNegative = bytes.length > 0 && (bytes[0] & 0x80) !== 0; - let value = 0n; - for (const byte of bytes) { - value = (value << 8n) | BigInt(byte); - } - // If negative, sign-extend: the bytes represent a two's complement value - if (isNegative) { - value = value - (1n << BigInt(bytes.length * 8)); - } + const value = twosComplementBytesToBigInt(bytes); - // Strip trailing zeros let exp = exponent; - while (value !== 0n && value % 10n === 0n) { - value = value / 10n; + let normalized = value; + while (normalized !== 0n && normalized % 10n === 0n) { + normalized = normalized / 10n; exp += 1; } - // If it fits in i64 range, use i64 encoding (more compact) - if (value >= -9223372036854775808n && value <= 9223372036854775807n) { - return { exponent: exp, mantissa: { type: "i64", value } }; + if (normalized >= -9223372036854775808n && normalized <= 9223372036854775807n) { + return { exponent: exp, mantissa: { type: "i64", value: normalized } }; } - // Convert back to big-endian two's complement bytes (minimal encoding) - const resultBytes = bigIntToMinimalTwosComplement(value); + if (normalized === value && exp === exponent) { + return { exponent, mantissa }; + } + + const resultBytes = bigIntToMinimalTwosComplement(normalized); return { exponent: exp, mantissa: { type: "big", bytes: resultBytes } }; } } @@ -245,7 +236,43 @@ function bigIntToMinimalTwosComplement(value: bigint): Uint8Array { encoded >>= 8n; } - return bytes; + return trimTwosComplement(bytes); +} + +function twosComplementBytesToBigInt(bytes: Uint8Array): bigint { + if (bytes.length === 0) { + return 0n; + } + + const isNegative = (bytes[0] & 0x80) !== 0; + let value = 0n; + for (const byte of bytes) { + value = (value << 8n) | BigInt(byte); + } + if (isNegative) { + value -= 1n << BigInt(bytes.length * 8); + } + return value; +} + +function hasRedundantSignExtension(bytes: Uint8Array): boolean { + return bytes.length > 1 + && ((bytes[0] === 0x00 && (bytes[1] & 0x80) === 0) + || (bytes[0] === 0xFF && (bytes[1] & 0x80) !== 0)); +} + +function trimTwosComplement(bytes: Uint8Array): Uint8Array { + let start = 0; + while (start < bytes.length - 1) { + const first = bytes[start]; + const second = bytes[start + 1]; + if ((first === 0x00 && (second & 0x80) === 0) || (first === 0xFF && (second & 0x80) !== 0)) { + start += 1; + continue; + } + break; + } + return start === 0 ? bytes : bytes.slice(start); } /** @@ -327,11 +354,15 @@ export function decodeValuePayload(reader: Reader, dataType: DataType): Value { if (mantissaType === 0x00) { mantissa = { type: "i64", value: reader.readSignedVarint() }; } else if (mantissaType === 0x01) { - mantissa = { type: "big", bytes: reader.readLengthPrefixedBytes() }; + const bytes = reader.readLengthPrefixedBytes(); + if (hasRedundantSignExtension(bytes)) { + throw new DecodeError("E005", "decimal mantissa bytes are not minimal"); + } + mantissa = { type: "big", bytes }; } else { throw new DecodeError("E005", `invalid decimal mantissa type: ${mantissaType}`); } - return { type: "decimal", exponent, mantissa }; + return { type: "decimal", ...normalizeDecimal(exponent, mantissa) }; } case DataType.Text: { diff --git a/typescript/src/test/basic.test.ts b/typescript/src/test/basic.test.ts index f52f0b3..7468201 100644 --- a/typescript/src/test/basic.test.ts +++ b/typescript/src/test/basic.test.ts @@ -34,7 +34,10 @@ import { languages, idsEqual, Reader, + Writer, } from "../index.js"; +import { decodeValuePayload, encodeValuePayload } from "../codec/value.js"; +import { DataType } from "../types/value.js"; function readObjectIds(encoded: Uint8Array): Id[] { const reader = new Reader(encoded); @@ -1471,6 +1474,58 @@ describe("Decimal normalization", () => { } } }); + + it("normalizes raw decoded decimals so TypeScript matches Rust", () => { + const writer = new Writer(); + writer.writeSignedVarint(-2n); + writer.writeByte(0x00); + writer.writeSignedVarint(100n); + + const decoded = decodeValuePayload(new Reader(writer.finish()), DataType.Decimal); + expect(decoded).toEqual({ + type: "decimal", + exponent: 0, + mantissa: { type: "i64", value: 1n }, + }); + }); + + it("normalizes raw decoded large big mantissas without truncation", () => { + const large = new Uint8Array([0x0A, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00]); + const writer = new Writer(); + writer.writeSignedVarint(0n); + writer.writeByte(0x01); + writer.writeLengthPrefixedBytes(large); + + const decoded = decodeValuePayload(new Reader(writer.finish()), DataType.Decimal); + expect(decoded).toEqual({ + type: "decimal", + exponent: 1, + mantissa: { + type: "big", + bytes: new Uint8Array([0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00]), + }, + }); + }); + + it("keeps negative big power-of-two mantissas minimally encoded", () => { + const writer = new Writer(); + encodeValuePayload(writer, { + type: "decimal", + exponent: 0, + mantissa: { + type: "big", + bytes: new Uint8Array([0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00]), + }, + }); + + const reader = new Reader(writer.finish()); + expect(reader.readSignedVarint()).toBe(0n); + expect(reader.readByte()).toBe(0x01); + const encodedBytes = reader.readLengthPrefixedBytes(); + expect(encodedBytes).toEqual(new Uint8Array([0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00])); + }); }); describe("Compression", () => { From 10884caaaf746b89373c85b59c82d79728b16d55 Mon Sep 17 00:00:00 2001 From: Nik Graf Date: Tue, 10 Mar 2026 09:01:02 +0100 Subject: [PATCH 3/9] Canonicalize decimal big mantissas across Rust and TypeScript MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tighten DECIMAL canonicalization so both runtimes always emit minimal two’s-complement mantissa bytes and reject redundant sign extension on decode. Update the Rust encoder to treat non-minimal big mantissas as requiring canonicalization, and rework the TypeScript big-mantissa path to normalize on bytes instead of preserving non-minimal input bytes. Add regression coverage for: - trimming non-minimal big mantissas on encode in Rust and TypeScript - rejecting non-minimal mantissa bytes on TypeScript decode - preserving the existing large-mantissa and negative boundary behavior This keeps Rust and TypeScript aligned and closes the latest PR review findings. --- rust/crates/grc-20/src/codec/value.rs | 43 +++++- typescript/src/codec/value.ts | 188 ++++++++++++++++++++++---- typescript/src/test/basic.test.ts | 29 ++++ 3 files changed, 232 insertions(+), 28 deletions(-) diff --git a/rust/crates/grc-20/src/codec/value.rs b/rust/crates/grc-20/src/codec/value.rs index 23ce185..3217d4c 100644 --- a/rust/crates/grc-20/src/codec/value.rs +++ b/rust/crates/grc-20/src/codec/value.rs @@ -168,6 +168,12 @@ fn is_big_mantissa_zero(bytes: &[u8]) -> bool { bytes.iter().all(|&b| b == 0) } +fn has_redundant_sign_extension(bytes: &[u8]) -> bool { + bytes.len() > 1 + && ((bytes[0] == 0x00 && (bytes[1] & 0x80) == 0) + || (bytes[0] == 0xFF && (bytes[1] & 0x80) != 0)) +} + /// Checks if a big-endian two's complement mantissa is divisible by 10. /// /// A number is divisible by 10 if its remainder when divided by 10 is 0. @@ -681,7 +687,8 @@ fn decimal_needs_canonicalization(exponent: i32, mantissa: &DecimalMantissa<'_>) match mantissa { DecimalMantissa::I64(v) => (*v == 0 && exponent != 0) || (*v != 0 && *v % 10 == 0), DecimalMantissa::Big(bytes) => { - is_big_mantissa_zero(bytes) + has_redundant_sign_extension(bytes) + || is_big_mantissa_zero(bytes) || is_big_mantissa_divisible_by_10(bytes) || twos_complement_bytes_to_i64(bytes).is_some() } @@ -1767,4 +1774,38 @@ mod tests { other => panic!("expected large normalized Decimal, got {other:?}"), } } + + #[test] + fn test_encode_decimal_trims_non_minimal_big_mantissa_bytes() { + let dicts = WireDictionaries::default(); + let mut dict_builder = DictionaryBuilder::new(); + let value = Value::Decimal { + exponent: 0, + mantissa: DecimalMantissa::Big(Cow::Owned(vec![ + 0x00, 0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF, 0x01, + ])), + unit: None, + }; + + let mut writer = Writer::new(); + encode_value(&mut writer, &value, &mut dict_builder).unwrap(); + + let mut reader = Reader::new(writer.as_bytes()); + let decoded = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap(); + + match decoded { + Value::Decimal { + exponent, + mantissa: DecimalMantissa::Big(bytes), + .. + } => { + assert_eq!(exponent, 0); + assert_eq!( + bytes.as_ref(), + &[0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF, 0x01] + ); + } + other => panic!("expected canonical big Decimal, got {other:?}"), + } + } } diff --git a/typescript/src/codec/value.ts b/typescript/src/codec/value.ts index b4c21f9..a18aaa5 100644 --- a/typescript/src/codec/value.ts +++ b/typescript/src/codec/value.ts @@ -181,29 +181,33 @@ function normalizeDecimal( } else { const bytes = mantissa.bytes; - if (bytes.every((b) => b === 0)) { + if (isBigMantissaZero(bytes)) { return { exponent: 0, mantissa: { type: "i64", value: 0n } }; } - const value = twosComplementBytesToBigInt(bytes); + if (!bigMantissaNeedsCanonicalization(exponent, bytes)) { + return { exponent, mantissa }; + } let exp = exponent; - let normalized = value; - while (normalized !== 0n && normalized % 10n === 0n) { - normalized = normalized / 10n; + const isNegative = (bytes[0] & 0x80) !== 0; + let magnitude = twosComplementAbsBytes(bytes); + while (true) { + const { quotient, remainder } = divideUnsignedBeBy10(magnitude); + if (remainder !== 0) { + break; + } + magnitude = quotient; exp += 1; } - if (normalized >= -9223372036854775808n && normalized <= 9223372036854775807n) { - return { exponent: exp, mantissa: { type: "i64", value: normalized } }; - } - - if (normalized === value && exp === exponent) { - return { exponent, mantissa }; + const canonicalBytes = signedBytesFromMagnitude(isNegative, magnitude); + const i64Value = twosComplementBytesToI64(canonicalBytes); + if (i64Value !== null) { + return { exponent: exp, mantissa: { type: "i64", value: i64Value } }; } - const resultBytes = bigIntToMinimalTwosComplement(normalized); - return { exponent: exp, mantissa: { type: "big", bytes: resultBytes } }; + return { exponent: exp, mantissa: { type: "big", bytes: canonicalBytes } }; } } @@ -239,20 +243,8 @@ function bigIntToMinimalTwosComplement(value: bigint): Uint8Array { return trimTwosComplement(bytes); } -function twosComplementBytesToBigInt(bytes: Uint8Array): bigint { - if (bytes.length === 0) { - return 0n; - } - - const isNegative = (bytes[0] & 0x80) !== 0; - let value = 0n; - for (const byte of bytes) { - value = (value << 8n) | BigInt(byte); - } - if (isNegative) { - value -= 1n << BigInt(bytes.length * 8); - } - return value; +function isBigMantissaZero(bytes: Uint8Array): boolean { + return bytes.every((byte) => byte === 0); } function hasRedundantSignExtension(bytes: Uint8Array): boolean { @@ -275,6 +267,148 @@ function trimTwosComplement(bytes: Uint8Array): Uint8Array { return start === 0 ? bytes : bytes.slice(start); } +function bigMantissaNeedsCanonicalization(exponent: number, bytes: Uint8Array): boolean { + return hasRedundantSignExtension(bytes) + || isBigMantissaZero(bytes) + || isBigMantissaDivisibleBy10(bytes) + || twosComplementBytesToI64(bytes) !== null; +} + +function isBigMantissaDivisibleBy10(bytes: Uint8Array): boolean { + if (bytes.length === 0) { + return true; + } + + const isNegative = (bytes[0] & 0x80) !== 0; + if (isNegative) { + return twosComplementAbsMod10(bytes) === 0; + } + + let remainder = 0; + for (const byte of bytes) { + remainder = (remainder * 6 + byte) % 10; + } + return remainder === 0; +} + +function twosComplementAbsMod10(bytes: Uint8Array): number { + let remainder = 0; + for (const byte of bytes) { + remainder = (remainder * 6 + (~byte & 0xFF)) % 10; + } + return (remainder + 1) % 10; +} + +function twosComplementAbsBytes(bytes: Uint8Array): Uint8Array { + if (bytes.length === 0) { + return new Uint8Array(); + } + + if ((bytes[0] & 0x80) === 0) { + let start = 0; + while (start < bytes.length && bytes[start] === 0) { + start += 1; + } + return bytes.slice(start); + } + + const magnitude = new Uint8Array(bytes.length); + let carry = 1; + for (let i = bytes.length - 1; i >= 0; i--) { + const sum = (~bytes[i] & 0xFF) + carry; + magnitude[i] = sum & 0xFF; + carry = sum >> 8; + } + + let start = 0; + while (start < magnitude.length && magnitude[start] === 0) { + start += 1; + } + return magnitude.slice(start); +} + +function divideUnsignedBeBy10(bytes: Uint8Array): { quotient: Uint8Array; remainder: number } { + const quotient = new Uint8Array(bytes.length); + let remainder = 0; + + for (let i = 0; i < bytes.length; i++) { + const current = (remainder << 8) | bytes[i]; + quotient[i] = Math.floor(current / 10); + remainder = current % 10; + } + + let start = 0; + while (start < quotient.length && quotient[start] === 0) { + start += 1; + } + return { quotient: quotient.slice(start), remainder }; +} + +function signedBytesFromMagnitude(isNegative: boolean, magnitude: Uint8Array): Uint8Array { + if (magnitude.length === 0) { + return new Uint8Array([0]); + } + + let start = 0; + while (start < magnitude.length && magnitude[start] === 0) { + start += 1; + } + const trimmed = magnitude.slice(start); + if (trimmed.length === 0) { + return new Uint8Array([0]); + } + + if (!isNegative) { + if ((trimmed[0] & 0x80) === 0) { + return trimmed; + } + const bytes = new Uint8Array(trimmed.length + 1); + bytes.set(trimmed, 1); + return bytes; + } + + const fitsInCurrentWidth = trimmed[0] < 0x80 + || (trimmed[0] === 0x80 && trimmed.slice(1).every((byte) => byte === 0)); + const width = fitsInCurrentWidth ? trimmed.length : trimmed.length + 1; + const bytes = new Uint8Array(width); + bytes.set(trimmed, width - trimmed.length); + + for (let i = 0; i < bytes.length; i++) { + bytes[i] = ~bytes[i] & 0xFF; + } + + let carry = 1; + for (let i = bytes.length - 1; i >= 0; i--) { + const sum = bytes[i] + carry; + bytes[i] = sum & 0xFF; + carry = sum >> 8; + } + + return trimTwosComplement(bytes); +} + +function twosComplementBytesToI64(bytes: Uint8Array): bigint | null { + if (bytes.length === 0) { + return 0n; + } + if (bytes.length > 8) { + return null; + } + + const fill = (bytes[0] & 0x80) !== 0 ? 0xFF : 0x00; + let value = 0n; + for (let i = 0; i < 8 - bytes.length; i++) { + value = (value << 8n) | BigInt(fill); + } + for (const byte of bytes) { + value = (value << 8n) | BigInt(byte); + } + if ((value & (1n << 63n)) !== 0n) { + value -= 1n << 64n; + } + return value; +} + /** * Encodes a decimal value. Normalizes the mantissa/exponent before encoding * to ensure trailing zeros are stripped. diff --git a/typescript/src/test/basic.test.ts b/typescript/src/test/basic.test.ts index 7468201..766a6d5 100644 --- a/typescript/src/test/basic.test.ts +++ b/typescript/src/test/basic.test.ts @@ -1526,6 +1526,35 @@ describe("Decimal normalization", () => { const encodedBytes = reader.readLengthPrefixedBytes(); expect(encodedBytes).toEqual(new Uint8Array([0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00])); }); + + it("re-encodes non-minimal big mantissas to canonical bytes", () => { + const writer = new Writer(); + encodeValuePayload(writer, { + type: "decimal", + exponent: 0, + mantissa: { + type: "big", + bytes: new Uint8Array([0x00, 0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF, 0x01]), + }, + }); + + const reader = new Reader(writer.finish()); + expect(reader.readSignedVarint()).toBe(0n); + expect(reader.readByte()).toBe(0x01); + expect(reader.readLengthPrefixedBytes()).toEqual( + new Uint8Array([0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF, 0x01]) + ); + }); + + it("rejects non-minimal decimal mantissa bytes on decode", () => { + const writer = new Writer(); + writer.writeSignedVarint(0n); + writer.writeByte(0x01); + writer.writeLengthPrefixedBytes(new Uint8Array([0x00, 0x7F])); + + expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) + .toThrow("decimal mantissa bytes are not minimal"); + }); }); describe("Compression", () => { From 6489fbf724231e344adea00a65d5530d5e1b5515 Mon Sep 17 00:00:00 2001 From: Nik Graf Date: Tue, 10 Mar 2026 09:08:17 +0100 Subject: [PATCH 4/9] Reject empty DECIMAL byte mantissas in Rust and TypeScript MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Treat mantissa_type = 0x01 with a zero-length payload as malformed instead of silently canonicalizing it to zero. This keeps empty byte mantissas distinct from the canonical zero representation and aligns both decoders with the minimal two’s-complement rule. Add regression tests in Rust and TypeScript to lock in rejection of empty big-mantissa payloads while preserving the existing decimal normalization behavior. --- rust/crates/grc-20/src/codec/value.rs | 37 +++++++++++++++++++-------- typescript/src/codec/value.ts | 2 +- typescript/src/test/basic.test.ts | 10 ++++++++ 3 files changed, 38 insertions(+), 11 deletions(-) diff --git a/rust/crates/grc-20/src/codec/value.rs b/rust/crates/grc-20/src/codec/value.rs index 3217d4c..5ed9488 100644 --- a/rust/crates/grc-20/src/codec/value.rs +++ b/rust/crates/grc-20/src/codec/value.rs @@ -114,17 +114,19 @@ fn decode_decimal<'a>( let len = reader.read_varint("decimal.mantissa_len")? as usize; let bytes = reader.read_bytes(len, "decimal.mantissa_bytes")?; + if bytes.is_empty() { + return Err(DecodeError::DecimalMantissaNotMinimal); + } + // Validate minimal encoding - if !bytes.is_empty() { - let first = bytes[0]; - // Check for redundant sign extension - if bytes.len() > 1 { - let second = bytes[1]; - if (first == 0x00 && (second & 0x80) == 0) - || (first == 0xFF && (second & 0x80) != 0) - { - return Err(DecodeError::DecimalMantissaNotMinimal); - } + let first = bytes[0]; + // Check for redundant sign extension + if bytes.len() > 1 { + let second = bytes[1]; + if (first == 0x00 && (second & 0x80) == 0) + || (first == 0xFF && (second & 0x80) != 0) + { + return Err(DecodeError::DecimalMantissaNotMinimal); } } @@ -1808,4 +1810,19 @@ mod tests { other => panic!("expected canonical big Decimal, got {other:?}"), } } + + #[test] + fn test_decode_decimal_rejects_empty_big_mantissa_bytes() { + let dicts = WireDictionaries::default(); + + let mut writer = Writer::new(); + writer.write_signed_varint(0); + writer.write_byte(0x01); + writer.write_varint(0); + writer.write_varint(0); + + let mut reader = Reader::new(writer.as_bytes()); + let err = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap_err(); + assert_eq!(err, DecodeError::DecimalMantissaNotMinimal); + } } diff --git a/typescript/src/codec/value.ts b/typescript/src/codec/value.ts index a18aaa5..66bd256 100644 --- a/typescript/src/codec/value.ts +++ b/typescript/src/codec/value.ts @@ -489,7 +489,7 @@ export function decodeValuePayload(reader: Reader, dataType: DataType): Value { mantissa = { type: "i64", value: reader.readSignedVarint() }; } else if (mantissaType === 0x01) { const bytes = reader.readLengthPrefixedBytes(); - if (hasRedundantSignExtension(bytes)) { + if (bytes.length === 0 || hasRedundantSignExtension(bytes)) { throw new DecodeError("E005", "decimal mantissa bytes are not minimal"); } mantissa = { type: "big", bytes }; diff --git a/typescript/src/test/basic.test.ts b/typescript/src/test/basic.test.ts index 766a6d5..d0d9c10 100644 --- a/typescript/src/test/basic.test.ts +++ b/typescript/src/test/basic.test.ts @@ -1555,6 +1555,16 @@ describe("Decimal normalization", () => { expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) .toThrow("decimal mantissa bytes are not minimal"); }); + + it("rejects empty big mantissa bytes on decode", () => { + const writer = new Writer(); + writer.writeSignedVarint(0n); + writer.writeByte(0x01); + writer.writeLengthPrefixedBytes(new Uint8Array([])); + + expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) + .toThrow("decimal mantissa bytes are not minimal"); + }); }); describe("Compression", () => { From 334f5a46fd73718d622ca730dafc585ad7ea6989 Mon Sep 17 00:00:00 2001 From: Nik Graf Date: Tue, 10 Mar 2026 09:11:52 +0100 Subject: [PATCH 5/9] Fix TypeScript build after decimal canonicalization rewrite Remove the now-unused BigInt mantissa helper and drop the unused exponent parameter from bigMantissaNeedsCanonicalization. These were left behind after moving TypeScript DECIMAL normalization to the byte-based path and caused tsc to fail in CI with TS6133 unused symbol errors. --- typescript/src/codec/value.ts | 36 ++--------------------------------- 1 file changed, 2 insertions(+), 34 deletions(-) diff --git a/typescript/src/codec/value.ts b/typescript/src/codec/value.ts index 66bd256..3b6bdc1 100644 --- a/typescript/src/codec/value.ts +++ b/typescript/src/codec/value.ts @@ -185,7 +185,7 @@ function normalizeDecimal( return { exponent: 0, mantissa: { type: "i64", value: 0n } }; } - if (!bigMantissaNeedsCanonicalization(exponent, bytes)) { + if (!bigMantissaNeedsCanonicalization(bytes)) { return { exponent, mantissa }; } @@ -211,38 +211,6 @@ function normalizeDecimal( } } -/** - * Converts a BigInt to minimal big-endian two's complement bytes. - */ -function bigIntToMinimalTwosComplement(value: bigint): Uint8Array { - if (value === 0n) { - return new Uint8Array([0]); - } - - const isNegative = value < 0n; - // Work with the absolute value to determine byte count, - // then encode in two's complement - const abs = isNegative ? -value : value; - - // Determine how many bytes we need - const bitLen = abs.toString(2).length; - // +1 for sign bit, then round up to bytes - const byteLen = Math.ceil((bitLen + 1) / 8); - - // Encode in two's complement - let encoded = isNegative - ? (1n << BigInt(byteLen * 8)) + value // two's complement for negative - : value; - - const bytes = new Uint8Array(byteLen); - for (let i = byteLen - 1; i >= 0; i--) { - bytes[i] = Number(encoded & 0xFFn); - encoded >>= 8n; - } - - return trimTwosComplement(bytes); -} - function isBigMantissaZero(bytes: Uint8Array): boolean { return bytes.every((byte) => byte === 0); } @@ -267,7 +235,7 @@ function trimTwosComplement(bytes: Uint8Array): Uint8Array { return start === 0 ? bytes : bytes.slice(start); } -function bigMantissaNeedsCanonicalization(exponent: number, bytes: Uint8Array): boolean { +function bigMantissaNeedsCanonicalization(bytes: Uint8Array): boolean { return hasRedundantSignExtension(bytes) || isBigMantissaZero(bytes) || isBigMantissaDivisibleBy10(bytes) From 411af3e986c8fb5157997745284d29f5ef8e8c6a Mon Sep 17 00:00:00 2001 From: Nik Graf Date: Tue, 10 Mar 2026 10:12:57 +0100 Subject: [PATCH 6/9] Harden decimal exponent and mantissa validation in Rust and TypeScript Reject DECIMAL exponents outside the int32 range in both decoders and fail normalization when stripping trailing zeros would overflow the exponent. This prevents silent truncation on decode and avoids debug panics or release-mode wrapping during canonicalization. Also enforce the shared 64 MB byte-length limit for big DECIMAL mantissas on decode so oversized payloads are rejected before normalization work begins. Add regression coverage for: - out-of-range DECIMAL exponents on decode - exponent overflow during encode-time and decode-time normalization - oversized big-mantissa byte payloads - the nested TypeScript decimal normalization test block formatting cleanup --- rust/crates/grc-20/src/codec/value.rs | 132 +++- typescript/src/codec/value.ts | 53 +- typescript/src/test/basic.test.ts | 1000 +++++++++++++------------ 3 files changed, 682 insertions(+), 503 deletions(-) diff --git a/rust/crates/grc-20/src/codec/value.rs b/rust/crates/grc-20/src/codec/value.rs index 5ed9488..04b2471 100644 --- a/rust/crates/grc-20/src/codec/value.rs +++ b/rust/crates/grc-20/src/codec/value.rs @@ -22,6 +22,8 @@ use crate::util::{ // DECODING // ============================================================================= +const DECIMAL_EXPONENT_OUT_OF_RANGE: &str = "DECIMAL exponent outside int32 range"; + /// Decodes a Value from the reader based on the data type (zero-copy). pub fn decode_value<'a>( reader: &mut Reader<'a>, @@ -102,7 +104,10 @@ fn decode_decimal<'a>( reader: &mut Reader<'a>, dicts: &WireDictionaries, ) -> Result, DecodeError> { - let exponent = reader.read_signed_varint("decimal.exponent")? as i32; + let exponent = i32::try_from(reader.read_signed_varint("decimal.exponent")?) + .map_err(|_| DecodeError::MalformedEncoding { + context: DECIMAL_EXPONENT_OUT_OF_RANGE, + })?; let mantissa_type = reader.read_byte("decimal.mantissa_type")?; let mantissa = match mantissa_type { @@ -112,6 +117,13 @@ fn decode_decimal<'a>( } 0x01 => { let len = reader.read_varint("decimal.mantissa_len")? as usize; + if len > MAX_BYTES_LEN { + return Err(DecodeError::LengthExceedsLimit { + field: "decimal.mantissa", + len, + max: MAX_BYTES_LEN, + }); + } let bytes = reader.read_bytes(len, "decimal.mantissa_bytes")?; if bytes.is_empty() { @@ -141,7 +153,7 @@ fn decode_decimal<'a>( // Normalize on read to handle already-published edits with non-normalized // decimals, while preserving borrowed bytes for canonical big mantissas. - let (exponent, mantissa) = normalize_decimal_for_decode(exponent, mantissa); + let (exponent, mantissa) = normalize_decimal_for_decode(exponent, mantissa)?; let unit_index = reader.read_varint("decimal.unit")? as usize; let unit = if unit_index == 0 { @@ -702,26 +714,26 @@ fn decimal_needs_canonicalization(exponent: i32, mantissa: &DecimalMantissa<'_>) fn normalize_decimal_owned( exponent: i32, mantissa: &DecimalMantissa<'_>, -) -> (i32, DecimalMantissa<'static>) { +) -> Option<(i32, DecimalMantissa<'static>)> { match mantissa { DecimalMantissa::I64(v) => { let mut value = *v; let mut exp = exponent; if value == 0 { - return (0, DecimalMantissa::I64(0)); + return Some((0, DecimalMantissa::I64(0))); } while value != 0 && value % 10 == 0 { value /= 10; - exp += 1; + exp = exp.checked_add(1)?; } - (exp, DecimalMantissa::I64(value)) + Some((exp, DecimalMantissa::I64(value))) } DecimalMantissa::Big(bytes) => { if is_big_mantissa_zero(bytes) { - return (0, DecimalMantissa::I64(0)); + return Some((0, DecimalMantissa::I64(0))); } let is_negative = bytes[0] & 0x80 != 0; @@ -734,14 +746,14 @@ fn normalize_decimal_owned( break; } magnitude = quotient; - exp += 1; + exp = exp.checked_add(1)?; } let canonical_bytes = signed_bytes_from_magnitude(is_negative, &magnitude); if let Some(value) = twos_complement_bytes_to_i64(&canonical_bytes) { - (exp, DecimalMantissa::I64(value)) + Some((exp, DecimalMantissa::I64(value))) } else { - (exp, DecimalMantissa::Big(Cow::Owned(canonical_bytes))) + Some((exp, DecimalMantissa::Big(Cow::Owned(canonical_bytes)))) } } } @@ -750,15 +762,20 @@ fn normalize_decimal_owned( fn normalize_decimal_for_decode<'a>( exponent: i32, mantissa: DecimalMantissa<'a>, -) -> (i32, DecimalMantissa<'a>) { +) -> Result<(i32, DecimalMantissa<'a>), DecodeError> { if !decimal_needs_canonicalization(exponent, &mantissa) { - return (exponent, mantissa); + return Ok((exponent, mantissa)); } - let (exp, canonical) = normalize_decimal_owned(exponent, &mantissa); + let (exp, canonical) = + normalize_decimal_owned(exponent, &mantissa).ok_or(DecodeError::MalformedEncoding { + context: DECIMAL_EXPONENT_OUT_OF_RANGE, + })?; match canonical { - DecimalMantissa::I64(value) => (exp, DecimalMantissa::I64(value)), - DecimalMantissa::Big(bytes) => (exp, DecimalMantissa::Big(Cow::Owned(bytes.into_owned()))), + DecimalMantissa::I64(value) => Ok((exp, DecimalMantissa::I64(value))), + DecimalMantissa::Big(bytes) => { + Ok((exp, DecimalMantissa::Big(Cow::Owned(bytes.into_owned())))) + } } } @@ -894,7 +911,10 @@ fn encode_decimal( return Ok(()); } - let (norm_exp, norm_mantissa) = normalize_decimal_owned(exponent, mantissa); + let (norm_exp, norm_mantissa) = + normalize_decimal_owned(exponent, mantissa).ok_or(EncodeError::InvalidInput { + context: DECIMAL_EXPONENT_OUT_OF_RANGE, + })?; writer.write_signed_varint(norm_exp as i64); match norm_mantissa { @@ -1825,4 +1845,84 @@ mod tests { let err = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap_err(); assert_eq!(err, DecodeError::DecimalMantissaNotMinimal); } + + #[test] + fn test_decode_decimal_rejects_exponent_outside_i32_range() { + let dicts = WireDictionaries::default(); + + let mut writer = Writer::new(); + writer.write_signed_varint(i32::MAX as i64 + 1); + writer.write_byte(0x00); + writer.write_signed_varint(1); + writer.write_varint(0); + + let mut reader = Reader::new(writer.as_bytes()); + let err = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap_err(); + assert_eq!( + err, + DecodeError::MalformedEncoding { + context: DECIMAL_EXPONENT_OUT_OF_RANGE, + } + ); + } + + #[test] + fn test_decode_decimal_rejects_exponent_overflow_during_normalization() { + let dicts = WireDictionaries::default(); + + let mut writer = Writer::new(); + writer.write_signed_varint(i32::MAX as i64); + writer.write_byte(0x00); + writer.write_signed_varint(10); + writer.write_varint(0); + + let mut reader = Reader::new(writer.as_bytes()); + let err = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap_err(); + assert_eq!( + err, + DecodeError::MalformedEncoding { + context: DECIMAL_EXPONENT_OUT_OF_RANGE, + } + ); + } + + #[test] + fn test_encode_decimal_rejects_exponent_overflow_during_normalization() { + let value = Value::Decimal { + exponent: i32::MAX, + mantissa: DecimalMantissa::I64(10), + unit: None, + }; + + let mut dict_builder = DictionaryBuilder::new(); + let mut writer = Writer::new(); + let err = encode_value(&mut writer, &value, &mut dict_builder).unwrap_err(); + assert_eq!( + err, + EncodeError::InvalidInput { + context: DECIMAL_EXPONENT_OUT_OF_RANGE, + } + ); + } + + #[test] + fn test_decode_decimal_rejects_big_mantissa_length_over_limit() { + let dicts = WireDictionaries::default(); + + let mut writer = Writer::new(); + writer.write_signed_varint(0); + writer.write_byte(0x01); + writer.write_varint(MAX_BYTES_LEN as u64 + 1); + + let mut reader = Reader::new(writer.as_bytes()); + let err = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap_err(); + assert_eq!( + err, + DecodeError::LengthExceedsLimit { + field: "decimal.mantissa", + len: MAX_BYTES_LEN + 1, + max: MAX_BYTES_LEN, + } + ); + } } diff --git a/typescript/src/codec/value.ts b/typescript/src/codec/value.ts index 3b6bdc1..8514b33 100644 --- a/typescript/src/codec/value.ts +++ b/typescript/src/codec/value.ts @@ -11,6 +11,11 @@ import { formatDatetimeRfc3339, } from "../util/datetime.js"; +const MIN_I32 = -2147483648; +const MAX_I32 = 2147483647; +const MAX_BYTES_LEN = 64 * 1024 * 1024; +const DECIMAL_EXPONENT_OUT_OF_RANGE = "DECIMAL exponent outside int32 range"; + /** * Dictionary builder for tracking property/language/unit indices. */ @@ -160,8 +165,11 @@ export function encodeValuePayload(writer: Writer, value: Value): void { */ function normalizeDecimal( exponent: number, - mantissa: DecimalMantissa + mantissa: DecimalMantissa, + mode: "encode" | "decode" ): { exponent: number; mantissa: DecimalMantissa } { + ensureDecimalExponent(exponent, mode); + if (mantissa.type === "i64") { let value = mantissa.value; let exp = exponent; @@ -174,7 +182,7 @@ function normalizeDecimal( // Strip trailing zeros while (value !== 0n && value % 10n === 0n) { value = value / 10n; - exp += 1; + exp = incrementDecimalExponent(exp, mode); } return { exponent: exp, mantissa: { type: "i64", value } }; @@ -198,7 +206,7 @@ function normalizeDecimal( break; } magnitude = quotient; - exp += 1; + exp = incrementDecimalExponent(exp, mode); } const canonicalBytes = signedBytesFromMagnitude(isNegative, magnitude); @@ -211,6 +219,33 @@ function normalizeDecimal( } } +function ensureDecimalExponent(exponent: number, mode: "encode" | "decode"): void { + if (!Number.isInteger(exponent) || exponent < MIN_I32 || exponent > MAX_I32) { + throw decimalExponentError(mode); + } +} + +function incrementDecimalExponent(exponent: number, mode: "encode" | "decode"): number { + if (exponent === MAX_I32) { + throw decimalExponentError(mode); + } + return exponent + 1; +} + +function decimalExponentError(mode: "encode" | "decode"): Error { + return mode === "decode" + ? new DecodeError("E005", DECIMAL_EXPONENT_OUT_OF_RANGE) + : new Error(DECIMAL_EXPONENT_OUT_OF_RANGE); +} + +function readDecimalExponent(reader: Reader): number { + const exponent = reader.readSignedVarint(); + if (exponent < BigInt(MIN_I32) || exponent > BigInt(MAX_I32)) { + throw decimalExponentError("decode"); + } + return Number(exponent); +} + function isBigMantissaZero(bytes: Uint8Array): boolean { return bytes.every((byte) => byte === 0); } @@ -382,7 +417,7 @@ function twosComplementBytesToI64(bytes: Uint8Array): bigint | null { * to ensure trailing zeros are stripped. */ function encodeDecimal(writer: Writer, exponent: number, mantissa: DecimalMantissa): void { - const normalized = normalizeDecimal(exponent, mantissa); + const normalized = normalizeDecimal(exponent, mantissa, "encode"); writer.writeSignedVarint(BigInt(normalized.exponent)); @@ -450,13 +485,17 @@ export function decodeValuePayload(reader: Reader, dataType: DataType): Value { } case DataType.Decimal: { - const exponent = Number(reader.readSignedVarint()); + const exponent = readDecimalExponent(reader); const mantissaType = reader.readByte(); let mantissa: DecimalMantissa; if (mantissaType === 0x00) { mantissa = { type: "i64", value: reader.readSignedVarint() }; } else if (mantissaType === 0x01) { - const bytes = reader.readLengthPrefixedBytes(); + const len = reader.readVarintNumber(); + if (len > MAX_BYTES_LEN) { + throw new DecodeError("E005", `decimal mantissa length ${len} exceeds maximum ${MAX_BYTES_LEN}`); + } + const bytes = new Uint8Array(reader.readBytes(len)); if (bytes.length === 0 || hasRedundantSignExtension(bytes)) { throw new DecodeError("E005", "decimal mantissa bytes are not minimal"); } @@ -464,7 +503,7 @@ export function decodeValuePayload(reader: Reader, dataType: DataType): Value { } else { throw new DecodeError("E005", `invalid decimal mantissa type: ${mantissaType}`); } - return { type: "decimal", ...normalizeDecimal(exponent, mantissa) }; + return { type: "decimal", ...normalizeDecimal(exponent, mantissa, "decode") }; } case DataType.Text: { diff --git a/typescript/src/test/basic.test.ts b/typescript/src/test/basic.test.ts index d0d9c10..81e30c1 100644 --- a/typescript/src/test/basic.test.ts +++ b/typescript/src/test/basic.test.ts @@ -1260,498 +1260,538 @@ describe("Codec", () => { expect(() => encodeEdit(edit)).toThrow("Invalid RFC 3339 time"); }); -describe("Decimal normalization", () => { - it("normalizes i64 mantissa with trailing zeros", () => { - const editId = randomId(); - const entityId = randomId(); - const propId = randomId(); - - // mantissa 100, exponent -2 represents 1.00 — should normalize to mantissa 1, exponent 0 - const edit: Edit = { - id: editId, - name: "Decimal Test", - authors: [], - createdAt: 0n, - ops: [ - { - type: "createEntity", - id: entityId, - values: [ - { property: propId, value: { type: "decimal", exponent: -2, mantissa: { type: "i64", value: 100n } } }, - ], - }, - ], - }; - - const encoded = encodeEdit(edit); - const decoded = decodeEdit(encoded); - - const op = decoded.ops[0]; - expect(op.type).toBe("createEntity"); - if (op.type === "createEntity") { - const val = op.values[0].value; - expect(val.type).toBe("decimal"); - if (val.type === "decimal") { - // Should be normalized: 100 * 10^-2 = 1 * 10^0 - expect(val.mantissa).toEqual({ type: "i64", value: 1n }); - expect(val.exponent).toBe(0); + describe("Decimal normalization", () => { + it("normalizes i64 mantissa with trailing zeros", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + // mantissa 100, exponent -2 represents 1.00 — should normalize to mantissa 1, exponent 0 + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: -2, mantissa: { type: "i64", value: 100n } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + expect(op.type).toBe("createEntity"); + if (op.type === "createEntity") { + const val = op.values[0].value; + expect(val.type).toBe("decimal"); + if (val.type === "decimal") { + // Should be normalized: 100 * 10^-2 = 1 * 10^0 + expect(val.mantissa).toEqual({ type: "i64", value: 1n }); + expect(val.exponent).toBe(0); + } } - } - }); - - it("normalizes i64 mantissa 1230 with exponent -2", () => { - const editId = randomId(); - const entityId = randomId(); - const propId = randomId(); - - const edit: Edit = { - id: editId, - name: "Decimal Test", - authors: [], - createdAt: 0n, - ops: [ - { - type: "createEntity", - id: entityId, - values: [ - { property: propId, value: { type: "decimal", exponent: -2, mantissa: { type: "i64", value: 1230n } } }, - ], - }, - ], - }; - - const encoded = encodeEdit(edit); - const decoded = decodeEdit(encoded); - - const op = decoded.ops[0]; - if (op.type === "createEntity") { - const val = op.values[0].value; - if (val.type === "decimal") { - // 1230 * 10^-2 = 123 * 10^-1 - expect(val.mantissa).toEqual({ type: "i64", value: 123n }); - expect(val.exponent).toBe(-1); + }); + + it("normalizes i64 mantissa 1230 with exponent -2", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: -2, mantissa: { type: "i64", value: 1230n } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + if (op.type === "createEntity") { + const val = op.values[0].value; + if (val.type === "decimal") { + // 1230 * 10^-2 = 123 * 10^-1 + expect(val.mantissa).toEqual({ type: "i64", value: 123n }); + expect(val.exponent).toBe(-1); + } } - } - }); - - it("normalizes zero mantissa with non-zero exponent", () => { - const editId = randomId(); - const entityId = randomId(); - const propId = randomId(); - - const edit: Edit = { - id: editId, - name: "Decimal Test", - authors: [], - createdAt: 0n, - ops: [ - { - type: "createEntity", - id: entityId, - values: [ - { property: propId, value: { type: "decimal", exponent: 5, mantissa: { type: "i64", value: 0n } } }, - ], - }, - ], - }; - - const encoded = encodeEdit(edit); - const decoded = decodeEdit(encoded); - - const op = decoded.ops[0]; - if (op.type === "createEntity") { - const val = op.values[0].value; - if (val.type === "decimal") { - expect(val.mantissa).toEqual({ type: "i64", value: 0n }); - expect(val.exponent).toBe(0); + }); + + it("normalizes zero mantissa with non-zero exponent", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: 5, mantissa: { type: "i64", value: 0n } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + if (op.type === "createEntity") { + const val = op.values[0].value; + if (val.type === "decimal") { + expect(val.mantissa).toEqual({ type: "i64", value: 0n }); + expect(val.exponent).toBe(0); + } } - } - }); - - it("does not modify already-normalized decimals", () => { - const editId = randomId(); - const entityId = randomId(); - const propId = randomId(); - - const edit: Edit = { - id: editId, - name: "Decimal Test", - authors: [], - createdAt: 0n, - ops: [ - { - type: "createEntity", - id: entityId, - values: [ - { property: propId, value: { type: "decimal", exponent: -2, mantissa: { type: "i64", value: 12345n } } }, - ], - }, - ], - }; - - const encoded = encodeEdit(edit); - const decoded = decodeEdit(encoded); - - const op = decoded.ops[0]; - if (op.type === "createEntity") { - const val = op.values[0].value; - if (val.type === "decimal") { - expect(val.mantissa).toEqual({ type: "i64", value: 12345n }); - expect(val.exponent).toBe(-2); + }); + + it("does not modify already-normalized decimals", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: -2, mantissa: { type: "i64", value: 12345n } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + if (op.type === "createEntity") { + const val = op.values[0].value; + if (val.type === "decimal") { + expect(val.mantissa).toEqual({ type: "i64", value: 12345n }); + expect(val.exponent).toBe(-2); + } } - } - }); - - it("normalizes negative mantissa with trailing zeros", () => { - const editId = randomId(); - const entityId = randomId(); - const propId = randomId(); - - const edit: Edit = { - id: editId, - name: "Decimal Test", - authors: [], - createdAt: 0n, - ops: [ - { - type: "createEntity", - id: entityId, - values: [ - { property: propId, value: { type: "decimal", exponent: 0, mantissa: { type: "i64", value: -500n } } }, - ], - }, - ], - }; - - const encoded = encodeEdit(edit); - const decoded = decodeEdit(encoded); - - const op = decoded.ops[0]; - if (op.type === "createEntity") { - const val = op.values[0].value; - if (val.type === "decimal") { - // -500 * 10^0 = -5 * 10^2 - expect(val.mantissa).toEqual({ type: "i64", value: -5n }); - expect(val.exponent).toBe(2); + }); + + it("normalizes negative mantissa with trailing zeros", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: 0, mantissa: { type: "i64", value: -500n } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + if (op.type === "createEntity") { + const val = op.values[0].value; + if (val.type === "decimal") { + // -500 * 10^0 = -5 * 10^2 + expect(val.mantissa).toEqual({ type: "i64", value: -5n }); + expect(val.exponent).toBe(2); + } } - } - }); - - it("normalizes big mantissa with trailing zeros", () => { - const editId = randomId(); - const entityId = randomId(); - const propId = randomId(); - - // Big-endian two's complement for 500: 0x01F4 - const bigBytes = new Uint8Array([0x01, 0xF4]); - - const edit: Edit = { - id: editId, - name: "Decimal Test", - authors: [], - createdAt: 0n, - ops: [ - { - type: "createEntity", - id: entityId, - values: [ - { property: propId, value: { type: "decimal", exponent: 0, mantissa: { type: "big", bytes: bigBytes } } }, - ], - }, - ], - }; - - const encoded = encodeEdit(edit); - const decoded = decodeEdit(encoded); - - const op = decoded.ops[0]; - if (op.type === "createEntity") { - const val = op.values[0].value; - if (val.type === "decimal") { - // 500 * 10^0 = 5 * 10^2, and 5 fits in i64 - expect(val.mantissa).toEqual({ type: "i64", value: 5n }); - expect(val.exponent).toBe(2); + }); + + it("normalizes big mantissa with trailing zeros", () => { + const editId = randomId(); + const entityId = randomId(); + const propId = randomId(); + + // Big-endian two's complement for 500: 0x01F4 + const bigBytes = new Uint8Array([0x01, 0xF4]); + + const edit: Edit = { + id: editId, + name: "Decimal Test", + authors: [], + createdAt: 0n, + ops: [ + { + type: "createEntity", + id: entityId, + values: [ + { property: propId, value: { type: "decimal", exponent: 0, mantissa: { type: "big", bytes: bigBytes } } }, + ], + }, + ], + }; + + const encoded = encodeEdit(edit); + const decoded = decodeEdit(encoded); + + const op = decoded.ops[0]; + if (op.type === "createEntity") { + const val = op.values[0].value; + if (val.type === "decimal") { + // 500 * 10^0 = 5 * 10^2, and 5 fits in i64 + expect(val.mantissa).toEqual({ type: "i64", value: 5n }); + expect(val.exponent).toBe(2); + } } - } - }); - - it("normalizes raw decoded decimals so TypeScript matches Rust", () => { - const writer = new Writer(); - writer.writeSignedVarint(-2n); - writer.writeByte(0x00); - writer.writeSignedVarint(100n); - - const decoded = decodeValuePayload(new Reader(writer.finish()), DataType.Decimal); - expect(decoded).toEqual({ - type: "decimal", - exponent: 0, - mantissa: { type: "i64", value: 1n }, }); - }); - - it("normalizes raw decoded large big mantissas without truncation", () => { - const large = new Uint8Array([0x0A, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, - 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00]); - const writer = new Writer(); - writer.writeSignedVarint(0n); - writer.writeByte(0x01); - writer.writeLengthPrefixedBytes(large); - - const decoded = decodeValuePayload(new Reader(writer.finish()), DataType.Decimal); - expect(decoded).toEqual({ - type: "decimal", - exponent: 1, - mantissa: { - type: "big", - bytes: new Uint8Array([0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, - 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00]), - }, + + it("normalizes raw decoded decimals so TypeScript matches Rust", () => { + const writer = new Writer(); + writer.writeSignedVarint(-2n); + writer.writeByte(0x00); + writer.writeSignedVarint(100n); + + const decoded = decodeValuePayload(new Reader(writer.finish()), DataType.Decimal); + expect(decoded).toEqual({ + type: "decimal", + exponent: 0, + mantissa: { type: "i64", value: 1n }, + }); }); - }); - - it("keeps negative big power-of-two mantissas minimally encoded", () => { - const writer = new Writer(); - encodeValuePayload(writer, { - type: "decimal", - exponent: 0, - mantissa: { - type: "big", - bytes: new Uint8Array([0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00]), - }, + + it("normalizes raw decoded large big mantissas without truncation", () => { + const large = new Uint8Array([0x0A, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00]); + const writer = new Writer(); + writer.writeSignedVarint(0n); + writer.writeByte(0x01); + writer.writeLengthPrefixedBytes(large); + + const decoded = decodeValuePayload(new Reader(writer.finish()), DataType.Decimal); + expect(decoded).toEqual({ + type: "decimal", + exponent: 1, + mantissa: { + type: "big", + bytes: new Uint8Array([0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00]), + }, + }); }); - - const reader = new Reader(writer.finish()); - expect(reader.readSignedVarint()).toBe(0n); - expect(reader.readByte()).toBe(0x01); - const encodedBytes = reader.readLengthPrefixedBytes(); - expect(encodedBytes).toEqual(new Uint8Array([0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00])); - }); - - it("re-encodes non-minimal big mantissas to canonical bytes", () => { - const writer = new Writer(); - encodeValuePayload(writer, { - type: "decimal", - exponent: 0, - mantissa: { - type: "big", - bytes: new Uint8Array([0x00, 0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF, 0x01]), - }, + + it("keeps negative big power-of-two mantissas minimally encoded", () => { + const writer = new Writer(); + encodeValuePayload(writer, { + type: "decimal", + exponent: 0, + mantissa: { + type: "big", + bytes: new Uint8Array([0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00]), + }, + }); + + const reader = new Reader(writer.finish()); + expect(reader.readSignedVarint()).toBe(0n); + expect(reader.readByte()).toBe(0x01); + const encodedBytes = reader.readLengthPrefixedBytes(); + expect(encodedBytes).toEqual(new Uint8Array([0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00])); }); - - const reader = new Reader(writer.finish()); - expect(reader.readSignedVarint()).toBe(0n); - expect(reader.readByte()).toBe(0x01); - expect(reader.readLengthPrefixedBytes()).toEqual( - new Uint8Array([0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF, 0x01]) - ); - }); - - it("rejects non-minimal decimal mantissa bytes on decode", () => { - const writer = new Writer(); - writer.writeSignedVarint(0n); - writer.writeByte(0x01); - writer.writeLengthPrefixedBytes(new Uint8Array([0x00, 0x7F])); - - expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) - .toThrow("decimal mantissa bytes are not minimal"); - }); - - it("rejects empty big mantissa bytes on decode", () => { - const writer = new Writer(); - writer.writeSignedVarint(0n); - writer.writeByte(0x01); - writer.writeLengthPrefixedBytes(new Uint8Array([])); - - expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) - .toThrow("decimal mantissa bytes are not minimal"); - }); -}); - -describe("Compression", () => { - it("isCompressed detects GRC2Z magic", () => { - const compressed = new Uint8Array([0x47, 0x52, 0x43, 0x32, 0x5a, 0x00]); // "GRC2Z" + data - const uncompressed = new Uint8Array([0x47, 0x52, 0x43, 0x32, 0x00]); // "GRC2" + data - - expect(isCompressed(compressed)).toBe(true); - expect(isCompressed(uncompressed)).toBe(false); - }); - - it("isCompressed returns false for short data", () => { - expect(isCompressed(new Uint8Array([0x47, 0x52, 0x43, 0x32]))).toBe(false); - expect(isCompressed(new Uint8Array([]))).toBe(false); - }); - - it("encodes and decodes compressed edit", async () => { - const editId = randomId(); - const entityId = randomId(); - - const edit = new EditBuilder(editId) - .setName("Compressed Test") - .setCreatedAt(1234567890000000n) - .createEntity(entityId, (e) => - e.text(properties.name(), "Alice", undefined) - .text(properties.description(), "A person named Alice with a long description to make compression worthwhile", undefined) - ) - .build(); - - const compressed = await encodeEditCompressed(edit); - - // Check magic bytes - expect(String.fromCharCode(...compressed.slice(0, 5))).toBe("GRC2Z"); - - // Verify it's detected as compressed - expect(isCompressed(compressed)).toBe(true); - - // Decode and verify - const decoded = await decodeEditCompressed(compressed); - - expect(idsEqual(decoded.id, edit.id)).toBe(true); - expect(decoded.name).toBe(edit.name); - expect(decoded.createdAt).toBe(edit.createdAt); - expect(decoded.ops.length).toBe(edit.ops.length); - }); - - it("compressed data is smaller than uncompressed for larger edits", async () => { - const editId = randomId(); - - // Create an edit with repetitive data (good for compression) - const builder = new EditBuilder(editId).setName("Large Test"); - - for (let i = 0; i < 50; i++) { - const entityId = randomId(); - builder.createEntity(entityId, (e) => - e.text(properties.name(), `Entity number ${i} with some padding text`, undefined) - .text(properties.description(), "This is a repeated description that should compress well", undefined) + + it("re-encodes non-minimal big mantissas to canonical bytes", () => { + const writer = new Writer(); + encodeValuePayload(writer, { + type: "decimal", + exponent: 0, + mantissa: { + type: "big", + bytes: new Uint8Array([0x00, 0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF, 0x01]), + }, + }); + + const reader = new Reader(writer.finish()); + expect(reader.readSignedVarint()).toBe(0n); + expect(reader.readByte()).toBe(0x01); + expect(reader.readLengthPrefixedBytes()).toEqual( + new Uint8Array([0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF, 0x01]) ); - } - - const edit = builder.build(); - - const uncompressed = encodeEdit(edit); - const compressed = await encodeEditCompressed(edit); - - // Compressed should be smaller - expect(compressed.length).toBeLessThan(uncompressed.length); - }); - - it("decodeEditAuto handles both formats", async () => { - const editId = randomId(); - const entityId = randomId(); - - const edit = new EditBuilder(editId) - .setName("Auto Test") - .createEntity(entityId, (e) => - e.text(properties.name(), "Test", undefined) - ) - .build(); - - const uncompressed = encodeEdit(edit); - const compressed = await encodeEditCompressed(edit); - - // Should decode both formats - const decoded1 = await decodeEditAuto(uncompressed); - const decoded2 = await decodeEditAuto(compressed); - - expect(idsEqual(decoded1.id, edit.id)).toBe(true); - expect(idsEqual(decoded2.id, edit.id)).toBe(true); - expect(decoded1.name).toBe(edit.name); - expect(decoded2.name).toBe(edit.name); - }); - - it("compressed canonical encoding roundtrips", async () => { - const editId = parseId("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa")!; - const entityId = parseId("bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb")!; - - const edit = new EditBuilder(editId) - .setName("Canonical Compressed Test") - .setCreatedAt(1000000n) - .createEntity(entityId, (e) => - e.text(properties.name(), "Test", undefined) - ) - .build(); - - const compressed = await encodeEditCompressed(edit, { canonical: true }); - const decoded = await decodeEditCompressed(compressed); - - expect(idsEqual(decoded.id, edit.id)).toBe(true); - expect(decoded.name).toBe(edit.name); - }); - - it("preloadCompression loads WASM", async () => { - // After preloading, compression should be ready - await preloadCompression(); - expect(isCompressionReady()).toBe(true); - }); - - it("encodeEditAuto returns uncompressed for small edits", async () => { - const editId = randomId(); - const entityId = randomId(); - - // Create a small edit - const edit = new EditBuilder(editId) - .setName("Small") - .createEmptyEntity(entityId) - .build(); - - // With default threshold, small edits should not be compressed - const encoded = await encodeEditAuto(edit); - - // Should have GRC2 magic (uncompressed) - expect(String.fromCharCode(...encoded.slice(0, 4))).toBe("GRC2"); - expect(isCompressed(encoded)).toBe(false); - - // Should decode correctly - const decoded = await decodeEditAuto(encoded); - expect(idsEqual(decoded.id, edit.id)).toBe(true); + }); + + it("rejects non-minimal decimal mantissa bytes on decode", () => { + const writer = new Writer(); + writer.writeSignedVarint(0n); + writer.writeByte(0x01); + writer.writeLengthPrefixedBytes(new Uint8Array([0x00, 0x7F])); + + expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) + .toThrow("decimal mantissa bytes are not minimal"); + }); + + it("rejects empty big mantissa bytes on decode", () => { + const writer = new Writer(); + writer.writeSignedVarint(0n); + writer.writeByte(0x01); + writer.writeLengthPrefixedBytes(new Uint8Array([])); + + expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) + .toThrow("decimal mantissa bytes are not minimal"); + }); + + it("rejects exponents outside int32 range on decode", () => { + const writer = new Writer(); + writer.writeSignedVarint(2147483648n); + writer.writeByte(0x00); + writer.writeSignedVarint(1n); + + expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) + .toThrow("DECIMAL exponent outside int32 range"); + }); + + it("rejects exponent overflow during decode-time normalization", () => { + const writer = new Writer(); + writer.writeSignedVarint(2147483647n); + writer.writeByte(0x00); + writer.writeSignedVarint(10n); + + expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) + .toThrow("DECIMAL exponent outside int32 range"); + }); + + it("rejects exponent overflow during encode-time normalization", () => { + const writer = new Writer(); + + expect(() => encodeValuePayload(writer, { + type: "decimal", + exponent: 2147483647, + mantissa: { type: "i64", value: 10n }, + })).toThrow("DECIMAL exponent outside int32 range"); + }); + + it("rejects decimal big mantissa lengths over the maximum", () => { + const writer = new Writer(); + writer.writeSignedVarint(0n); + writer.writeByte(0x01); + writer.writeVarintNumber(64 * 1024 * 1024 + 1); + + expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) + .toThrow("decimal mantissa length 67108865 exceeds maximum 67108864"); + }); }); - it("encodeEditAuto compresses large edits", async () => { - const editId = randomId(); - - // Create a large edit - const builder = new EditBuilder(editId).setName("Large Auto Test"); - for (let i = 0; i < 20; i++) { + describe("Compression", () => { + it("isCompressed detects GRC2Z magic", () => { + const compressed = new Uint8Array([0x47, 0x52, 0x43, 0x32, 0x5a, 0x00]); // "GRC2Z" + data + const uncompressed = new Uint8Array([0x47, 0x52, 0x43, 0x32, 0x00]); // "GRC2" + data + + expect(isCompressed(compressed)).toBe(true); + expect(isCompressed(uncompressed)).toBe(false); + }); + + it("isCompressed returns false for short data", () => { + expect(isCompressed(new Uint8Array([0x47, 0x52, 0x43, 0x32]))).toBe(false); + expect(isCompressed(new Uint8Array([]))).toBe(false); + }); + + it("encodes and decodes compressed edit", async () => { + const editId = randomId(); const entityId = randomId(); - builder.createEntity(entityId, (e) => - e.text(properties.name(), `Entity ${i} with padding`, undefined) - .text(properties.description(), "Repeated description for compression", undefined) - ); - } - const edit = builder.build(); - - // Should be compressed (above default threshold) - const encoded = await encodeEditAuto(edit); - expect(isCompressed(encoded)).toBe(true); - - // Should decode correctly - const decoded = await decodeEditAuto(encoded); - expect(idsEqual(decoded.id, edit.id)).toBe(true); - expect(decoded.ops.length).toBe(edit.ops.length); - }); - - it("encodeEditAuto respects threshold option", async () => { - const editId = randomId(); - const entityId = randomId(); - - const edit = new EditBuilder(editId) - .setName("Threshold Test") - .createEntity(entityId, (e) => - e.text(properties.name(), "Test", undefined) - ) - .build(); - - // With threshold: 0, should always compress - const alwaysCompressed = await encodeEditAuto(edit, { threshold: 0 }); - expect(isCompressed(alwaysCompressed)).toBe(true); - - // With threshold: Infinity, should never compress - const neverCompressed = await encodeEditAuto(edit, { threshold: Infinity }); - expect(isCompressed(neverCompressed)).toBe(false); - - // Both should decode correctly - const decoded1 = await decodeEditAuto(alwaysCompressed); - const decoded2 = await decodeEditAuto(neverCompressed); - expect(idsEqual(decoded1.id, edit.id)).toBe(true); - expect(idsEqual(decoded2.id, edit.id)).toBe(true); + + const edit = new EditBuilder(editId) + .setName("Compressed Test") + .setCreatedAt(1234567890000000n) + .createEntity(entityId, (e) => + e.text(properties.name(), "Alice", undefined) + .text(properties.description(), "A person named Alice with a long description to make compression worthwhile", undefined) + ) + .build(); + + const compressed = await encodeEditCompressed(edit); + + // Check magic bytes + expect(String.fromCharCode(...compressed.slice(0, 5))).toBe("GRC2Z"); + + // Verify it's detected as compressed + expect(isCompressed(compressed)).toBe(true); + + // Decode and verify + const decoded = await decodeEditCompressed(compressed); + + expect(idsEqual(decoded.id, edit.id)).toBe(true); + expect(decoded.name).toBe(edit.name); + expect(decoded.createdAt).toBe(edit.createdAt); + expect(decoded.ops.length).toBe(edit.ops.length); + }); + + it("compressed data is smaller than uncompressed for larger edits", async () => { + const editId = randomId(); + + // Create an edit with repetitive data (good for compression) + const builder = new EditBuilder(editId).setName("Large Test"); + + for (let i = 0; i < 50; i++) { + const entityId = randomId(); + builder.createEntity(entityId, (e) => + e.text(properties.name(), `Entity number ${i} with some padding text`, undefined) + .text(properties.description(), "This is a repeated description that should compress well", undefined) + ); + } + + const edit = builder.build(); + + const uncompressed = encodeEdit(edit); + const compressed = await encodeEditCompressed(edit); + + // Compressed should be smaller + expect(compressed.length).toBeLessThan(uncompressed.length); + }); + + it("decodeEditAuto handles both formats", async () => { + const editId = randomId(); + const entityId = randomId(); + + const edit = new EditBuilder(editId) + .setName("Auto Test") + .createEntity(entityId, (e) => + e.text(properties.name(), "Test", undefined) + ) + .build(); + + const uncompressed = encodeEdit(edit); + const compressed = await encodeEditCompressed(edit); + + // Should decode both formats + const decoded1 = await decodeEditAuto(uncompressed); + const decoded2 = await decodeEditAuto(compressed); + + expect(idsEqual(decoded1.id, edit.id)).toBe(true); + expect(idsEqual(decoded2.id, edit.id)).toBe(true); + expect(decoded1.name).toBe(edit.name); + expect(decoded2.name).toBe(edit.name); + }); + + it("compressed canonical encoding roundtrips", async () => { + const editId = parseId("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa")!; + const entityId = parseId("bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb")!; + + const edit = new EditBuilder(editId) + .setName("Canonical Compressed Test") + .setCreatedAt(1000000n) + .createEntity(entityId, (e) => + e.text(properties.name(), "Test", undefined) + ) + .build(); + + const compressed = await encodeEditCompressed(edit, { canonical: true }); + const decoded = await decodeEditCompressed(compressed); + + expect(idsEqual(decoded.id, edit.id)).toBe(true); + expect(decoded.name).toBe(edit.name); + }); + + it("preloadCompression loads WASM", async () => { + // After preloading, compression should be ready + await preloadCompression(); + expect(isCompressionReady()).toBe(true); + }); + + it("encodeEditAuto returns uncompressed for small edits", async () => { + const editId = randomId(); + const entityId = randomId(); + + // Create a small edit + const edit = new EditBuilder(editId) + .setName("Small") + .createEmptyEntity(entityId) + .build(); + + // With default threshold, small edits should not be compressed + const encoded = await encodeEditAuto(edit); + + // Should have GRC2 magic (uncompressed) + expect(String.fromCharCode(...encoded.slice(0, 4))).toBe("GRC2"); + expect(isCompressed(encoded)).toBe(false); + + // Should decode correctly + const decoded = await decodeEditAuto(encoded); + expect(idsEqual(decoded.id, edit.id)).toBe(true); + }); + + it("encodeEditAuto compresses large edits", async () => { + const editId = randomId(); + + // Create a large edit + const builder = new EditBuilder(editId).setName("Large Auto Test"); + for (let i = 0; i < 20; i++) { + const entityId = randomId(); + builder.createEntity(entityId, (e) => + e.text(properties.name(), `Entity ${i} with padding`, undefined) + .text(properties.description(), "Repeated description for compression", undefined) + ); + } + const edit = builder.build(); + + // Should be compressed (above default threshold) + const encoded = await encodeEditAuto(edit); + expect(isCompressed(encoded)).toBe(true); + + // Should decode correctly + const decoded = await decodeEditAuto(encoded); + expect(idsEqual(decoded.id, edit.id)).toBe(true); + expect(decoded.ops.length).toBe(edit.ops.length); + }); + + it("encodeEditAuto respects threshold option", async () => { + const editId = randomId(); + const entityId = randomId(); + + const edit = new EditBuilder(editId) + .setName("Threshold Test") + .createEntity(entityId, (e) => + e.text(properties.name(), "Test", undefined) + ) + .build(); + + // With threshold: 0, should always compress + const alwaysCompressed = await encodeEditAuto(edit, { threshold: 0 }); + expect(isCompressed(alwaysCompressed)).toBe(true); + + // With threshold: Infinity, should never compress + const neverCompressed = await encodeEditAuto(edit, { threshold: Infinity }); + expect(isCompressed(neverCompressed)).toBe(false); + + // Both should decode correctly + const decoded1 = await decodeEditAuto(alwaysCompressed); + const decoded2 = await decodeEditAuto(neverCompressed); + expect(idsEqual(decoded1.id, edit.id)).toBe(true); + expect(idsEqual(decoded2.id, edit.id)).toBe(true); + }); }); -}); From bc0e9216cae590bf6ddda542f6a76061af6d03ed Mon Sep 17 00:00:00 2001 From: Nik Graf Date: Tue, 10 Mar 2026 10:29:21 +0100 Subject: [PATCH 7/9] Reject empty decimal big mantissas on encode Treat empty DECIMAL big-mantissa byte arrays as invalid input instead of silently canonicalizing them to zero. This aligns encode-time behavior with the existing decode-time minimality checks and keeps Rust and TypeScript consistent. Also tighten the zero-detection path so empty byte arrays are no longer treated as numeric zero, and avoid an extra full-array scan in the TypeScript big- mantissa canonicalization path. Add regression coverage for rejecting empty big mantissas during encoding in both runtimes, and update the Rust helper test to reflect the stricter invariant. --- rust/crates/grc-20/src/codec/value.rs | 88 +++++++++++++++++++++------ typescript/src/codec/value.ts | 21 +++++-- typescript/src/test/basic.test.ts | 14 ++++- 3 files changed, 99 insertions(+), 24 deletions(-) diff --git a/rust/crates/grc-20/src/codec/value.rs b/rust/crates/grc-20/src/codec/value.rs index 04b2471..ab2e52f 100644 --- a/rust/crates/grc-20/src/codec/value.rs +++ b/rust/crates/grc-20/src/codec/value.rs @@ -23,6 +23,37 @@ use crate::util::{ // ============================================================================= const DECIMAL_EXPONENT_OUT_OF_RANGE: &str = "DECIMAL exponent outside int32 range"; +const DECIMAL_EMPTY_BIG_MANTISSA: &str = "DECIMAL mantissa bytes must not be empty"; + +#[derive(Clone, Copy)] +enum DecimalNormalizeError { + ExponentOutOfRange, + EmptyBigMantissa, +} + +impl DecimalNormalizeError { + fn into_decode_error(self) -> DecodeError { + match self { + DecimalNormalizeError::ExponentOutOfRange => DecodeError::MalformedEncoding { + context: DECIMAL_EXPONENT_OUT_OF_RANGE, + }, + DecimalNormalizeError::EmptyBigMantissa => DecodeError::MalformedEncoding { + context: DECIMAL_EMPTY_BIG_MANTISSA, + }, + } + } + + fn into_encode_error(self) -> EncodeError { + match self { + DecimalNormalizeError::ExponentOutOfRange => EncodeError::InvalidInput { + context: DECIMAL_EXPONENT_OUT_OF_RANGE, + }, + DecimalNormalizeError::EmptyBigMantissa => EncodeError::InvalidInput { + context: DECIMAL_EMPTY_BIG_MANTISSA, + }, + } + } +} /// Decodes a Value from the reader based on the data type (zero-copy). pub fn decode_value<'a>( @@ -179,7 +210,7 @@ fn decode_decimal<'a>( /// Checks if a big-endian two's complement mantissa represents zero. fn is_big_mantissa_zero(bytes: &[u8]) -> bool { - bytes.iter().all(|&b| b == 0) + !bytes.is_empty() && bytes.iter().all(|&b| b == 0) } fn has_redundant_sign_extension(bytes: &[u8]) -> bool { @@ -714,26 +745,32 @@ fn decimal_needs_canonicalization(exponent: i32, mantissa: &DecimalMantissa<'_>) fn normalize_decimal_owned( exponent: i32, mantissa: &DecimalMantissa<'_>, -) -> Option<(i32, DecimalMantissa<'static>)> { +) -> Result<(i32, DecimalMantissa<'static>), DecimalNormalizeError> { match mantissa { DecimalMantissa::I64(v) => { let mut value = *v; let mut exp = exponent; if value == 0 { - return Some((0, DecimalMantissa::I64(0))); + return Ok((0, DecimalMantissa::I64(0))); } while value != 0 && value % 10 == 0 { value /= 10; - exp = exp.checked_add(1)?; + exp = exp + .checked_add(1) + .ok_or(DecimalNormalizeError::ExponentOutOfRange)?; } - Some((exp, DecimalMantissa::I64(value))) + Ok((exp, DecimalMantissa::I64(value))) } DecimalMantissa::Big(bytes) => { + if bytes.is_empty() { + return Err(DecimalNormalizeError::EmptyBigMantissa); + } + if is_big_mantissa_zero(bytes) { - return Some((0, DecimalMantissa::I64(0))); + return Ok((0, DecimalMantissa::I64(0))); } let is_negative = bytes[0] & 0x80 != 0; @@ -746,14 +783,16 @@ fn normalize_decimal_owned( break; } magnitude = quotient; - exp = exp.checked_add(1)?; + exp = exp + .checked_add(1) + .ok_or(DecimalNormalizeError::ExponentOutOfRange)?; } let canonical_bytes = signed_bytes_from_magnitude(is_negative, &magnitude); if let Some(value) = twos_complement_bytes_to_i64(&canonical_bytes) { - Some((exp, DecimalMantissa::I64(value))) + Ok((exp, DecimalMantissa::I64(value))) } else { - Some((exp, DecimalMantissa::Big(Cow::Owned(canonical_bytes)))) + Ok((exp, DecimalMantissa::Big(Cow::Owned(canonical_bytes)))) } } } @@ -767,10 +806,8 @@ fn normalize_decimal_for_decode<'a>( return Ok((exponent, mantissa)); } - let (exp, canonical) = - normalize_decimal_owned(exponent, &mantissa).ok_or(DecodeError::MalformedEncoding { - context: DECIMAL_EXPONENT_OUT_OF_RANGE, - })?; + let (exp, canonical) = normalize_decimal_owned(exponent, &mantissa) + .map_err(DecimalNormalizeError::into_decode_error)?; match canonical { DecimalMantissa::I64(value) => Ok((exp, DecimalMantissa::I64(value))), DecimalMantissa::Big(bytes) => { @@ -912,9 +949,7 @@ fn encode_decimal( } let (norm_exp, norm_mantissa) = - normalize_decimal_owned(exponent, mantissa).ok_or(EncodeError::InvalidInput { - context: DECIMAL_EXPONENT_OUT_OF_RANGE, - })?; + normalize_decimal_owned(exponent, mantissa).map_err(DecimalNormalizeError::into_encode_error)?; writer.write_signed_varint(norm_exp as i64); match norm_mantissa { @@ -1578,7 +1613,7 @@ mod tests { #[test] fn test_big_decimal_normalization_helpers() { // Test is_big_mantissa_zero - assert!(is_big_mantissa_zero(&[])); + assert!(!is_big_mantissa_zero(&[])); assert!(is_big_mantissa_zero(&[0])); assert!(is_big_mantissa_zero(&[0, 0, 0])); assert!(!is_big_mantissa_zero(&[1])); @@ -1905,6 +1940,25 @@ mod tests { ); } + #[test] + fn test_encode_decimal_rejects_empty_big_mantissa_bytes() { + let value = Value::Decimal { + exponent: 0, + mantissa: DecimalMantissa::Big(Cow::Owned(Vec::new())), + unit: None, + }; + + let mut dict_builder = DictionaryBuilder::new(); + let mut writer = Writer::new(); + let err = encode_value(&mut writer, &value, &mut dict_builder).unwrap_err(); + assert_eq!( + err, + EncodeError::InvalidInput { + context: DECIMAL_EMPTY_BIG_MANTISSA, + } + ); + } + #[test] fn test_decode_decimal_rejects_big_mantissa_length_over_limit() { let dicts = WireDictionaries::default(); diff --git a/typescript/src/codec/value.ts b/typescript/src/codec/value.ts index 8514b33..b7ad369 100644 --- a/typescript/src/codec/value.ts +++ b/typescript/src/codec/value.ts @@ -15,6 +15,7 @@ const MIN_I32 = -2147483648; const MAX_I32 = 2147483647; const MAX_BYTES_LEN = 64 * 1024 * 1024; const DECIMAL_EXPONENT_OUT_OF_RANGE = "DECIMAL exponent outside int32 range"; +const DECIMAL_EMPTY_BIG_MANTISSA = "DECIMAL mantissa bytes must not be empty"; /** * Dictionary builder for tracking property/language/unit indices. @@ -188,12 +189,16 @@ function normalizeDecimal( return { exponent: exp, mantissa: { type: "i64", value } }; } else { const bytes = mantissa.bytes; + if (bytes.length === 0) { + throw decimalEmptyBigMantissaError(mode); + } - if (isBigMantissaZero(bytes)) { + const isZero = isBigMantissaZero(bytes); + if (isZero) { return { exponent: 0, mantissa: { type: "i64", value: 0n } }; } - if (!bigMantissaNeedsCanonicalization(bytes)) { + if (!bigMantissaNeedsCanonicalization(bytes, isZero)) { return { exponent, mantissa }; } @@ -238,6 +243,12 @@ function decimalExponentError(mode: "encode" | "decode"): Error { : new Error(DECIMAL_EXPONENT_OUT_OF_RANGE); } +function decimalEmptyBigMantissaError(mode: "encode" | "decode"): Error { + return mode === "decode" + ? new DecodeError("E005", DECIMAL_EMPTY_BIG_MANTISSA) + : new Error(DECIMAL_EMPTY_BIG_MANTISSA); +} + function readDecimalExponent(reader: Reader): number { const exponent = reader.readSignedVarint(); if (exponent < BigInt(MIN_I32) || exponent > BigInt(MAX_I32)) { @@ -247,7 +258,7 @@ function readDecimalExponent(reader: Reader): number { } function isBigMantissaZero(bytes: Uint8Array): boolean { - return bytes.every((byte) => byte === 0); + return bytes.length > 0 && bytes.every((byte) => byte === 0); } function hasRedundantSignExtension(bytes: Uint8Array): boolean { @@ -270,9 +281,9 @@ function trimTwosComplement(bytes: Uint8Array): Uint8Array { return start === 0 ? bytes : bytes.slice(start); } -function bigMantissaNeedsCanonicalization(bytes: Uint8Array): boolean { +function bigMantissaNeedsCanonicalization(bytes: Uint8Array, isZero: boolean = isBigMantissaZero(bytes)): boolean { return hasRedundantSignExtension(bytes) - || isBigMantissaZero(bytes) + || isZero || isBigMantissaDivisibleBy10(bytes) || twosComplementBytesToI64(bytes) !== null; } diff --git a/typescript/src/test/basic.test.ts b/typescript/src/test/basic.test.ts index 81e30c1..b908b72 100644 --- a/typescript/src/test/basic.test.ts +++ b/typescript/src/test/basic.test.ts @@ -1561,11 +1561,21 @@ describe("Codec", () => { writer.writeSignedVarint(0n); writer.writeByte(0x01); writer.writeLengthPrefixedBytes(new Uint8Array([])); - + expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) .toThrow("decimal mantissa bytes are not minimal"); }); - + + it("rejects empty big mantissa bytes on encode", () => { + const writer = new Writer(); + + expect(() => encodeValuePayload(writer, { + type: "decimal", + exponent: 0, + mantissa: { type: "big", bytes: new Uint8Array([]) }, + })).toThrow("DECIMAL mantissa bytes must not be empty"); + }); + it("rejects exponents outside int32 range on decode", () => { const writer = new Writer(); writer.writeSignedVarint(2147483648n); From 6279db6923e738f8e2d2363b3e12c220d8555978 Mon Sep 17 00:00:00 2001 From: Nik Graf Date: Tue, 10 Mar 2026 11:13:38 +0100 Subject: [PATCH 8/9] Bound decimal normalization work and align encode-time limits Enforce the DECIMAL big-mantissa byte limit on encode in both Rust and TypeScript so the library cannot emit payloads that its own decoder rejects. Also cap big-mantissa normalization to a fixed number of divide-by-10 steps to prevent pathological legacy decimals with huge trailing base-10 factors from causing excessive CPU work during encode-time or decode-time normalization. Add regression coverage for: - rejecting oversized big mantissas on encode - rejecting decimals that exceed the normalization step limit on encode/decode - preserving the existing aligned Rust/TypeScript DECIMAL behavior --- rust/crates/grc-20/src/codec/value.rs | 118 ++++++++++++++++++++++++++ typescript/src/codec/value.ts | 30 ++++++- typescript/src/test/basic.test.ts | 56 ++++++++++++ 3 files changed, 203 insertions(+), 1 deletion(-) diff --git a/rust/crates/grc-20/src/codec/value.rs b/rust/crates/grc-20/src/codec/value.rs index ab2e52f..5a77a1e 100644 --- a/rust/crates/grc-20/src/codec/value.rs +++ b/rust/crates/grc-20/src/codec/value.rs @@ -24,11 +24,15 @@ use crate::util::{ const DECIMAL_EXPONENT_OUT_OF_RANGE: &str = "DECIMAL exponent outside int32 range"; const DECIMAL_EMPTY_BIG_MANTISSA: &str = "DECIMAL mantissa bytes must not be empty"; +const DECIMAL_NORMALIZATION_STEP_LIMIT_EXCEEDED: &str = + "DECIMAL normalization exceeds maximum steps"; +const MAX_DECIMAL_NORMALIZATION_STEPS: usize = 4096; #[derive(Clone, Copy)] enum DecimalNormalizeError { ExponentOutOfRange, EmptyBigMantissa, + StepLimitExceeded, } impl DecimalNormalizeError { @@ -40,6 +44,9 @@ impl DecimalNormalizeError { DecimalNormalizeError::EmptyBigMantissa => DecodeError::MalformedEncoding { context: DECIMAL_EMPTY_BIG_MANTISSA, }, + DecimalNormalizeError::StepLimitExceeded => DecodeError::MalformedEncoding { + context: DECIMAL_NORMALIZATION_STEP_LIMIT_EXCEEDED, + }, } } @@ -51,6 +58,9 @@ impl DecimalNormalizeError { DecimalNormalizeError::EmptyBigMantissa => EncodeError::InvalidInput { context: DECIMAL_EMPTY_BIG_MANTISSA, }, + DecimalNormalizeError::StepLimitExceeded => EncodeError::InvalidInput { + context: DECIMAL_NORMALIZATION_STEP_LIMIT_EXCEEDED, + }, } } } @@ -776,12 +786,17 @@ fn normalize_decimal_owned( let is_negative = bytes[0] & 0x80 != 0; let mut magnitude = twos_complement_abs_bytes(bytes); let mut exp = exponent; + let mut steps = 0usize; loop { let (quotient, remainder) = divide_unsigned_be_by_10(&magnitude); if remainder != 0 { break; } + steps += 1; + if steps > MAX_DECIMAL_NORMALIZATION_STEPS { + return Err(DecimalNormalizeError::StepLimitExceeded); + } magnitude = quotient; exp = exp .checked_add(1) @@ -930,6 +945,16 @@ fn encode_decimal( exponent: i32, mantissa: &DecimalMantissa<'_>, ) -> Result<(), EncodeError> { + if let DecimalMantissa::Big(bytes) = mantissa { + if bytes.len() > MAX_BYTES_LEN { + return Err(EncodeError::LengthExceedsLimit { + field: "decimal.mantissa", + len: bytes.len(), + max: MAX_BYTES_LEN, + }); + } + } + // Normalize: strip trailing zeros from mantissa, adjust exponent, and use // the compact i64 wire form whenever the canonical mantissa fits in i64. if !decimal_needs_canonicalization(exponent, mantissa) { @@ -1642,6 +1667,33 @@ mod tests { assert!(!is_big_mantissa_divisible_by_10(&[0xF9])); // -7 } + fn multiply_unsigned_be_by_10(bytes: &[u8]) -> Vec { + let mut result = Vec::with_capacity(bytes.len() + 1); + let mut carry = 0u16; + + for &byte in bytes.iter().rev() { + let product = byte as u16 * 10 + carry; + result.push((product & 0xFF) as u8); + carry = product >> 8; + } + + while carry != 0 { + result.push((carry & 0xFF) as u8); + carry >>= 8; + } + + result.reverse(); + result + } + + fn decimal_power_of_ten_bytes(power: usize) -> Vec { + let mut magnitude = vec![1u8]; + for _ in 0..power { + magnitude = multiply_unsigned_be_by_10(&magnitude); + } + signed_bytes_from_magnitude(false, &magnitude) + } + #[test] fn test_big_decimal_normalization_encode() { let dicts = WireDictionaries::default(); @@ -1959,6 +2011,72 @@ mod tests { ); } + #[test] + fn test_encode_decimal_rejects_big_mantissa_length_over_limit() { + let mut oversized = vec![0u8; MAX_BYTES_LEN + 1]; + oversized[0] = 1; + let value = Value::Decimal { + exponent: 0, + mantissa: DecimalMantissa::Big(Cow::Owned(oversized)), + unit: None, + }; + + let mut dict_builder = DictionaryBuilder::new(); + let mut writer = Writer::new(); + let err = encode_value(&mut writer, &value, &mut dict_builder).unwrap_err(); + assert_eq!( + err, + EncodeError::LengthExceedsLimit { + field: "decimal.mantissa", + len: MAX_BYTES_LEN + 1, + max: MAX_BYTES_LEN, + } + ); + } + + #[test] + fn test_decode_decimal_rejects_excessive_normalization_steps() { + let dicts = WireDictionaries::default(); + let bytes = decimal_power_of_ten_bytes(MAX_DECIMAL_NORMALIZATION_STEPS + 1); + + let mut writer = Writer::new(); + writer.write_signed_varint(0); + writer.write_byte(0x01); + writer.write_varint(bytes.len() as u64); + writer.write_bytes(&bytes); + writer.write_varint(0); + + let mut reader = Reader::new(writer.as_bytes()); + let err = decode_value(&mut reader, DataType::Decimal, &dicts).unwrap_err(); + assert_eq!( + err, + DecodeError::MalformedEncoding { + context: DECIMAL_NORMALIZATION_STEP_LIMIT_EXCEEDED, + } + ); + } + + #[test] + fn test_encode_decimal_rejects_excessive_normalization_steps() { + let value = Value::Decimal { + exponent: 0, + mantissa: DecimalMantissa::Big(Cow::Owned(decimal_power_of_ten_bytes( + MAX_DECIMAL_NORMALIZATION_STEPS + 1, + ))), + unit: None, + }; + + let mut dict_builder = DictionaryBuilder::new(); + let mut writer = Writer::new(); + let err = encode_value(&mut writer, &value, &mut dict_builder).unwrap_err(); + assert_eq!( + err, + EncodeError::InvalidInput { + context: DECIMAL_NORMALIZATION_STEP_LIMIT_EXCEEDED, + } + ); + } + #[test] fn test_decode_decimal_rejects_big_mantissa_length_over_limit() { let dicts = WireDictionaries::default(); diff --git a/typescript/src/codec/value.ts b/typescript/src/codec/value.ts index b7ad369..3bc9c63 100644 --- a/typescript/src/codec/value.ts +++ b/typescript/src/codec/value.ts @@ -16,6 +16,8 @@ const MAX_I32 = 2147483647; const MAX_BYTES_LEN = 64 * 1024 * 1024; const DECIMAL_EXPONENT_OUT_OF_RANGE = "DECIMAL exponent outside int32 range"; const DECIMAL_EMPTY_BIG_MANTISSA = "DECIMAL mantissa bytes must not be empty"; +const DECIMAL_NORMALIZATION_STEP_LIMIT_EXCEEDED = "DECIMAL normalization exceeds maximum steps"; +const MAX_DECIMAL_NORMALIZATION_STEPS = 4096; /** * Dictionary builder for tracking property/language/unit indices. @@ -170,6 +172,7 @@ function normalizeDecimal( mode: "encode" | "decode" ): { exponent: number; mantissa: DecimalMantissa } { ensureDecimalExponent(exponent, mode); + ensureDecimalBigMantissaLength(mantissa, mode); if (mantissa.type === "i64") { let value = mantissa.value; @@ -205,11 +208,16 @@ function normalizeDecimal( let exp = exponent; const isNegative = (bytes[0] & 0x80) !== 0; let magnitude = twosComplementAbsBytes(bytes); + let steps = 0; while (true) { const { quotient, remainder } = divideUnsignedBeBy10(magnitude); if (remainder !== 0) { break; } + steps += 1; + if (steps > MAX_DECIMAL_NORMALIZATION_STEPS) { + throw decimalNormalizationStepLimitError(mode); + } magnitude = quotient; exp = incrementDecimalExponent(exp, mode); } @@ -249,6 +257,26 @@ function decimalEmptyBigMantissaError(mode: "encode" | "decode"): Error { : new Error(DECIMAL_EMPTY_BIG_MANTISSA); } +function decimalNormalizationStepLimitError(mode: "encode" | "decode"): Error { + return mode === "decode" + ? new DecodeError("E005", DECIMAL_NORMALIZATION_STEP_LIMIT_EXCEEDED) + : new Error(DECIMAL_NORMALIZATION_STEP_LIMIT_EXCEEDED); +} + +function ensureDecimalBigMantissaLength( + mantissa: DecimalMantissa, + mode: "encode" | "decode" +): void { + if (mantissa.type === "big" && mantissa.bytes.length > MAX_BYTES_LEN) { + throw decimalBigMantissaLengthError(mantissa.bytes.length, mode); + } +} + +function decimalBigMantissaLengthError(length: number, mode: "encode" | "decode"): Error { + const message = `decimal mantissa length ${length} exceeds maximum ${MAX_BYTES_LEN}`; + return mode === "decode" ? new DecodeError("E005", message) : new Error(message); +} + function readDecimalExponent(reader: Reader): number { const exponent = reader.readSignedVarint(); if (exponent < BigInt(MIN_I32) || exponent > BigInt(MAX_I32)) { @@ -504,7 +532,7 @@ export function decodeValuePayload(reader: Reader, dataType: DataType): Value { } else if (mantissaType === 0x01) { const len = reader.readVarintNumber(); if (len > MAX_BYTES_LEN) { - throw new DecodeError("E005", `decimal mantissa length ${len} exceeds maximum ${MAX_BYTES_LEN}`); + throw decimalBigMantissaLengthError(len, "decode"); } const bytes = new Uint8Array(reader.readBytes(len)); if (bytes.length === 0 || hasRedundantSignExtension(bytes)) { diff --git a/typescript/src/test/basic.test.ts b/typescript/src/test/basic.test.ts index b908b72..7bad2d1 100644 --- a/typescript/src/test/basic.test.ts +++ b/typescript/src/test/basic.test.ts @@ -60,6 +60,29 @@ function readObjectIds(encoded: Uint8Array): Id[] { return objects; } +function positiveBigIntToMinimalTwosComplement(value: bigint): Uint8Array { + let hex = value.toString(16); + if (hex.length % 2 !== 0) { + hex = `0${hex}`; + } + + const raw = new Uint8Array(hex.length / 2); + for (let i = 0; i < raw.length; i++) { + raw[i] = parseInt(hex.slice(i * 2, i * 2 + 2), 16); + } + + if (raw.length === 0) { + return new Uint8Array([0]); + } + if ((raw[0] & 0x80) === 0) { + return raw; + } + + const bytes = new Uint8Array(raw.length + 1); + bytes.set(raw, 1); + return bytes; +} + describe("ID utilities", () => { it("formatId produces 32 hex chars", () => { const id = randomId(); @@ -1576,6 +1599,39 @@ describe("Codec", () => { })).toThrow("DECIMAL mantissa bytes must not be empty"); }); + it("rejects decimal big mantissas over the maximum on encode", () => { + const writer = new Writer(); + const oversized = new Uint8Array(64 * 1024 * 1024 + 1); + oversized[0] = 1; + + expect(() => encodeValuePayload(writer, { + type: "decimal", + exponent: 0, + mantissa: { type: "big", bytes: oversized }, + })).toThrow("decimal mantissa length 67108865 exceeds maximum 67108864"); + }); + + it("rejects excessive decimal normalization work on decode", () => { + const writer = new Writer(); + const bytes = positiveBigIntToMinimalTwosComplement(10n ** 4097n); + writer.writeSignedVarint(0n); + writer.writeByte(0x01); + writer.writeLengthPrefixedBytes(bytes); + + expect(() => decodeValuePayload(new Reader(writer.finish()), DataType.Decimal)) + .toThrow("DECIMAL normalization exceeds maximum steps"); + }); + + it("rejects excessive decimal normalization work on encode", () => { + const writer = new Writer(); + + expect(() => encodeValuePayload(writer, { + type: "decimal", + exponent: 0, + mantissa: { type: "big", bytes: positiveBigIntToMinimalTwosComplement(10n ** 4097n) }, + })).toThrow("DECIMAL normalization exceeds maximum steps"); + }); + it("rejects exponents outside int32 range on decode", () => { const writer = new Writer(); writer.writeSignedVarint(2147483648n); From 19793c956f9a998eb058c8d34dc64778ac301748 Mon Sep 17 00:00:00 2001 From: Nik Graf Date: Tue, 10 Mar 2026 12:28:21 +0100 Subject: [PATCH 9/9] add changeset --- .../.changeset/decimal-validation-and-normalization.md | 7 +++++++ 1 file changed, 7 insertions(+) create mode 100644 typescript/.changeset/decimal-validation-and-normalization.md diff --git a/typescript/.changeset/decimal-validation-and-normalization.md b/typescript/.changeset/decimal-validation-and-normalization.md new file mode 100644 index 0000000..f87ba35 --- /dev/null +++ b/typescript/.changeset/decimal-validation-and-normalization.md @@ -0,0 +1,7 @@ +--- +"@geoprotocol/grc-20": patch +--- + +Harden decimal encoding and decoding in the TypeScript codec. + +Decimal values are now normalized canonically during encode and decode, with better validation for exponent bounds, empty or non-minimal big mantissas, oversized mantissa byte lengths, and decimal inputs that would require excessive normalization work.