Skip to main content

iceberg/spec/values/
decimal_utils.rs

1// Licensed to the Apache Software Foundation (ASF) under one
2// or more contributor license agreements.  See the NOTICE file
3// distributed with this work for additional information
4// regarding copyright ownership.  The ASF licenses this file
5// to you under the Apache License, Version 2.0 (the
6// "License"); you may not use this file except in compliance
7// with the License.  You may obtain a copy of the License at
8//
9//   http://www.apache.org/licenses/LICENSE-2.0
10//
11// Unless required by applicable law or agreed to in writing,
12// software distributed under the License is distributed on an
13// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
14// KIND, either express or implied.  See the License for the
15// specific language governing permissions and limitations
16// under the License.
17
18//! Compatibility layer for decimal operations.
19//!
20//! Provides rust_decimal-compatible API using fastnum's D128 internally.
21//! D128 supports 38-digit precision, meeting the Iceberg spec requirement.
22
23use fastnum::D128;
24use fastnum::decimal::Context;
25
26use crate::Result;
27use crate::error::invalid_data;
28
29/// Re-export D128 as the Decimal type for use throughout the crate.
30pub type Decimal = D128;
31
32/// Create a D128 from mantissa (i128) and scale (u32).
33///
34/// This is equivalent to rust_decimal's `Decimal::from_i128_with_scale`.
35/// The value is computed as: mantissa * 10^(-scale)
36///
37/// For example:
38/// - mantissa=12345, scale=2 => 123.45
39/// - mantissa=-456, scale=3 => -0.456
40pub fn decimal_from_i128_with_scale(mantissa: i128, scale: u32) -> Decimal {
41    if scale == 0 {
42        return D128::from_i128(mantissa).expect("i128 always fits in D128");
43    }
44
45    // Convert mantissa to string and insert decimal point at the right position
46    let is_negative = mantissa < 0;
47    let abs_str = mantissa.unsigned_abs().to_string();
48    let scale_usize = scale as usize;
49
50    let decimal_str = if abs_str.len() <= scale_usize {
51        // Need leading zeros: e.g., mantissa=456, scale=3 => "0.456"
52        // Or mantissa=5, scale=3 => "0.005"
53        let zeros_needed = scale_usize - abs_str.len();
54        format!(
55            "{}0.{}{}",
56            if is_negative { "-" } else { "" },
57            "0".repeat(zeros_needed),
58            abs_str
59        )
60    } else {
61        // Insert decimal point: e.g., mantissa=12345, scale=2 => "123.45"
62        let decimal_pos = abs_str.len() - scale_usize;
63        format!(
64            "{}{}.{}",
65            if is_negative { "-" } else { "" },
66            &abs_str[..decimal_pos],
67            &abs_str[decimal_pos..]
68        )
69    };
70
71    D128::from_str(&decimal_str, Context::default())
72        .expect("constructed decimal string is always valid")
73}
74
75/// Try to create a D128 from mantissa and scale, with validation.
76///
77/// This is equivalent to rust_decimal's `Decimal::try_from_i128_with_scale`.
78/// Currently always succeeds for 38-digit decimals.
79pub fn try_decimal_from_i128_with_scale(mantissa: i128, scale: u32) -> Result<Decimal> {
80    // For now, always succeeds since D128 supports full 38-digit precision
81    Ok(decimal_from_i128_with_scale(mantissa, scale))
82}
83
84/// Create a D128 from i64 mantissa and scale.
85///
86/// This is equivalent to rust_decimal's `Decimal::new`.
87#[allow(dead_code)]
88pub fn decimal_new(mantissa: i64, scale: u32) -> Decimal {
89    decimal_from_i128_with_scale(mantissa as i128, scale)
90}
91
92/// Parse a decimal from string with exact representation.
93///
94/// This is equivalent to rust_decimal's `Decimal::from_str_exact`.
95pub fn decimal_from_str_exact(s: &str) -> Result<Decimal> {
96    D128::from_str(s, Context::default()).map_err(|e| invalid_data!("Can't parse decimal: {e}"))
97}
98
99/// Get the mantissa (unscaled coefficient) as i128.
100///
101/// This is equivalent to rust_decimal's `decimal.mantissa()`.
102///
103/// The mantissa is signed: negative decimals return negative mantissa.
104pub fn decimal_mantissa(d: &Decimal) -> i128 {
105    // digits() returns unsigned coefficient as UInt<N>
106    // For Iceberg decimals (max 38 digits), this always fits in u128/i128
107    let digits = d.digits();
108
109    // Convert UInt<2> to u128 - this always succeeds for Iceberg-compliant decimals
110    // since 38 digits requires ~127 bits and u128 has 128 bits
111    let unsigned: u128 = digits
112        .to_u128()
113        .expect("Iceberg decimals (max 38 digits) always fit in u128");
114
115    let signed = unsigned as i128;
116    if d.is_sign_negative() {
117        -signed
118    } else {
119        signed
120    }
121}
122
123/// Get the decimal digit precision of a mantissa.
124///
125/// A mantissa of 0 has a precision of 1.
126pub fn decimal_precision(mantissa: i128) -> u32 {
127    mantissa
128        .unsigned_abs()
129        .checked_ilog10()
130        .map_or(1, |x| x + 1)
131}
132
133/// Get the scale (number of digits after decimal point).
134///
135/// This is equivalent to rust_decimal's `decimal.scale()`.
136pub fn decimal_scale(d: &Decimal) -> u32 {
137    let frac = d.fractional_digits_count();
138    if frac < 0 { 0 } else { frac as u32 }
139}
140
141/// Rescale a decimal to the given scale, returning the rescaled value.
142///
143/// This is equivalent to rust_decimal's `decimal.rescale(scale)`.
144pub fn decimal_rescale(d: Decimal, scale: u32) -> Decimal {
145    d.rescale(scale as i16)
146}
147
148/// Convert big-endian signed bytes to i128.
149///
150/// This handles variable-length byte arrays (up to 16 bytes) with sign extension.
151/// Returns None if the byte array is longer than 16 bytes.
152pub fn i128_from_be_bytes(bytes: &[u8]) -> Option<i128> {
153    if bytes.is_empty() {
154        return Some(0);
155    }
156    if bytes.len() > 16 {
157        return None; // Too large for i128
158    }
159
160    // Check sign bit (most significant bit of first byte)
161    let is_negative = bytes[0] & 0x80 != 0;
162
163    // Pad to 16 bytes with sign extension
164    let mut padded = if is_negative { [0xFF; 16] } else { [0; 16] };
165    let start = 16 - bytes.len();
166    padded[start..].copy_from_slice(bytes);
167
168    Some(i128::from_be_bytes(padded))
169}
170
171/// Convert i128 to big-endian signed bytes with minimum length.
172///
173/// This produces the shortest two's complement representation of the value.
174/// The result is suitable for Iceberg decimal binary serialization.
175pub fn i128_to_be_bytes_min(value: i128) -> Vec<u8> {
176    let bytes = value.to_be_bytes();
177
178    // Find the first significant byte
179    // For positive numbers, skip leading 0x00 bytes (but keep sign bit)
180    // For negative numbers, skip leading 0xFF bytes (but keep sign bit)
181    let is_negative = value < 0;
182    let skip_byte = if is_negative { 0xFF } else { 0x00 };
183
184    let mut start = 0;
185    while start < 15 && bytes[start] == skip_byte {
186        // Check if the next byte has the correct sign bit
187        let next_byte = bytes[start + 1];
188        let next_is_negative = (next_byte & 0x80) != 0;
189        if next_is_negative == is_negative {
190            start += 1;
191        } else {
192            break;
193        }
194    }
195
196    bytes[start..].to_vec()
197}
198
199/// Encode an i128 as exactly `len` big-endian two's complement bytes, matching a
200/// Parquet `FIXED_LEN_BYTE_ARRAY` column of that declared `type_length`.
201///
202/// Returns `None` if `value` does not fit in `len` bytes, since a truncated
203/// encoding would represent a different number.
204pub(crate) fn decimal_to_fixed_length_bytes_exact(value: i128, len: usize) -> Option<Vec<u8>> {
205    if len == 0 || len > 16 {
206        return None;
207    }
208
209    let be_bytes = value.to_be_bytes();
210    let offset = 16 - len;
211    let sign_byte = if value < 0 { 0xFF } else { 0x00 };
212
213    // The bytes being dropped must be pure sign extension.
214    if be_bytes[..offset].iter().any(|&b| b != sign_byte) {
215        return None;
216    }
217
218    // The retained leading byte must carry the value's sign.
219    if (be_bytes[offset] & 0x80 != 0) != (value < 0) {
220        return None;
221    }
222
223    Some(be_bytes[offset..].to_vec())
224}
225
226#[cfg(test)]
227mod tests {
228    use super::*;
229
230    #[test]
231    fn test_decimal_from_i128_with_scale() {
232        let d = decimal_from_i128_with_scale(12345, 2);
233        assert_eq!(d.to_string(), "123.45");
234
235        let d = decimal_from_i128_with_scale(-12345, 2);
236        assert_eq!(d.to_string(), "-123.45");
237
238        let d = decimal_from_i128_with_scale(0, 5);
239        assert_eq!(d.to_string(), "0.00000");
240    }
241
242    #[test]
243    fn test_decimal_new() {
244        let d = decimal_new(123, 2);
245        assert_eq!(d.to_string(), "1.23");
246
247        let d = decimal_new(-456, 3);
248        assert_eq!(d.to_string(), "-0.456");
249    }
250
251    #[test]
252    fn test_decimal_from_str_exact() {
253        let d = decimal_from_str_exact("123.45").unwrap();
254        assert_eq!(d.to_string(), "123.45");
255
256        let d = decimal_from_str_exact("-0.001").unwrap();
257        assert_eq!(d.to_string(), "-0.001");
258
259        let d = decimal_from_str_exact("99999999999999999999999999999999999999").unwrap();
260        assert_eq!(d.to_string(), "99999999999999999999999999999999999999");
261    }
262
263    #[test]
264    fn test_decimal_mantissa() {
265        let d = decimal_from_i128_with_scale(12345, 2);
266        assert_eq!(decimal_mantissa(&d), 12345);
267
268        let d = decimal_from_i128_with_scale(-12345, 2);
269        assert_eq!(decimal_mantissa(&d), -12345);
270    }
271
272    #[test]
273    fn test_decimal_precision() {
274        assert_eq!(decimal_precision(0), 1);
275        assert_eq!(decimal_precision(5), 1);
276        assert_eq!(decimal_precision(42), 2);
277        assert_eq!(decimal_precision(-42), 2);
278        // power-of-10 boundaries, where off-by-one bugs in log10 logic live
279        assert_eq!(decimal_precision(9), 1);
280        assert_eq!(decimal_precision(10), 2);
281        assert_eq!(decimal_precision(99), 2);
282        assert_eq!(decimal_precision(100), 3);
283
284        // max Iceberg decimal precision
285        assert_eq!(
286            decimal_precision(99999999999999999999999999999999999999),
287            38
288        );
289        assert_eq!(
290            decimal_precision(-99999999999999999999999999999999999999),
291            38
292        );
293
294        // i128 extremes
295        assert_eq!(decimal_precision(i128::MAX), 39);
296        assert_eq!(decimal_precision(i128::MIN), 39);
297    }
298
299    #[test]
300    fn test_decimal_scale() {
301        let d = decimal_from_i128_with_scale(12345, 2);
302        assert_eq!(decimal_scale(&d), 2);
303
304        let d = decimal_from_i128_with_scale(12345, 0);
305        assert_eq!(decimal_scale(&d), 0);
306    }
307
308    #[test]
309    fn test_decimal_rescale() {
310        let d = decimal_from_str_exact("123.45").unwrap();
311        let rescaled = decimal_rescale(d, 4);
312        assert_eq!(decimal_scale(&rescaled), 4);
313        assert_eq!(decimal_mantissa(&rescaled), 1234500);
314    }
315
316    #[test]
317    fn test_38_digit_precision() {
318        // Test that we can handle 38-digit decimals (Iceberg spec requirement)
319        let max_38_digits = "99999999999999999999999999999999999999";
320        let d = decimal_from_str_exact(max_38_digits).unwrap();
321        assert_eq!(d.to_string(), max_38_digits);
322
323        let min_38_digits = "-99999999999999999999999999999999999999";
324        let d = decimal_from_str_exact(min_38_digits).unwrap();
325        assert_eq!(d.to_string(), min_38_digits);
326    }
327
328    #[test]
329    fn test_i128_from_be_bytes() {
330        // Empty bytes
331        assert_eq!(i128_from_be_bytes(&[]), Some(0));
332
333        // Positive values
334        assert_eq!(i128_from_be_bytes(&[0x01]), Some(1));
335        assert_eq!(i128_from_be_bytes(&[0x7F]), Some(127));
336        assert_eq!(i128_from_be_bytes(&[0x00, 0xFF]), Some(255));
337        assert_eq!(i128_from_be_bytes(&[0x04, 0xD2]), Some(1234));
338
339        // Negative values (sign extension)
340        assert_eq!(i128_from_be_bytes(&[0xFF]), Some(-1));
341        assert_eq!(i128_from_be_bytes(&[0x80]), Some(-128));
342        assert_eq!(i128_from_be_bytes(&[0xFB, 0x2E]), Some(-1234));
343
344        // Too large (> 16 bytes)
345        assert_eq!(i128_from_be_bytes(&[0; 17]), None);
346    }
347
348    #[test]
349    fn test_i128_to_be_bytes_min() {
350        // Positive values
351        assert_eq!(i128_to_be_bytes_min(0), vec![0x00]);
352        assert_eq!(i128_to_be_bytes_min(1), vec![0x01]);
353        assert_eq!(i128_to_be_bytes_min(127), vec![0x7F]);
354        assert_eq!(i128_to_be_bytes_min(128), vec![0x00, 0x80]);
355        assert_eq!(i128_to_be_bytes_min(255), vec![0x00, 0xFF]);
356        assert_eq!(i128_to_be_bytes_min(1234), vec![0x04, 0xD2]);
357
358        // Negative values
359        assert_eq!(i128_to_be_bytes_min(-1), vec![0xFF]);
360        assert_eq!(i128_to_be_bytes_min(-128), vec![0x80]);
361        assert_eq!(i128_to_be_bytes_min(-129), vec![0xFF, 0x7F]);
362        assert_eq!(i128_to_be_bytes_min(-1234), vec![0xFB, 0x2E]);
363
364        // Round trip test
365        for val in [
366            0i128,
367            1,
368            -1,
369            127,
370            -128,
371            255,
372            -256,
373            12345,
374            -12345,
375            i128::MAX,
376            i128::MIN,
377        ] {
378            let bytes = i128_to_be_bytes_min(val);
379            assert_eq!(
380                i128_from_be_bytes(&bytes),
381                Some(val),
382                "Round trip failed for {val}"
383            );
384        }
385    }
386
387    #[test]
388    fn test_decimal_to_fixed_length_bytes_exact_positive() {
389        let bytes = decimal_to_fixed_length_bytes_exact(12345, 9).unwrap();
390        assert_eq!(bytes.len(), 9);
391        // Big-endian, zero-padded on the left: 12345 = 0x3039
392        assert_eq!(bytes[7], 0x30);
393        assert_eq!(bytes[8], 0x39);
394        assert!(bytes[..7].iter().all(|&b| b == 0x00));
395    }
396
397    #[test]
398    fn test_decimal_to_fixed_length_bytes_exact_negative() {
399        let bytes = decimal_to_fixed_length_bytes_exact(-1, 9).unwrap();
400        assert_eq!(bytes.len(), 9);
401        assert!(bytes.iter().all(|&b| b == 0xFF));
402    }
403
404    #[test]
405    fn test_decimal_to_fixed_length_bytes_exact_round_trip() {
406        for value in [
407            0i128,
408            1,
409            -1,
410            12345,
411            -12345,
412            i64::MAX as i128,
413            i64::MIN as i128,
414        ] {
415            let bytes = decimal_to_fixed_length_bytes_exact(value, 16).unwrap();
416            assert_eq!(bytes.len(), 16);
417            assert_eq!(
418                i128_from_be_bytes(&bytes),
419                Some(value),
420                "Round trip failed for value={value}"
421            );
422        }
423    }
424
425    /// The length must be honoured exactly: a value needing more bytes cannot be
426    /// truncated into the column's width, because the truncation would encode a
427    /// different number and probe the wrong bloom filter slot.
428    #[test]
429    fn test_decimal_to_fixed_length_bytes_exact_rejects_overflow() {
430        // 200 needs a leading zero byte to stay positive in two's complement.
431        assert_eq!(decimal_to_fixed_length_bytes_exact(200, 1), None);
432        assert_eq!(
433            decimal_to_fixed_length_bytes_exact(200, 2),
434            Some(vec![0x00, 0xC8])
435        );
436
437        assert_eq!(decimal_to_fixed_length_bytes_exact(-200, 1), None);
438        assert_eq!(
439            decimal_to_fixed_length_bytes_exact(-200, 2),
440            Some(vec![0xFF, 0x38])
441        );
442
443        // Exactly representable boundaries.
444        assert_eq!(
445            decimal_to_fixed_length_bytes_exact(127, 1),
446            Some(vec![0x7F])
447        );
448        assert_eq!(decimal_to_fixed_length_bytes_exact(128, 1), None);
449        assert_eq!(
450            decimal_to_fixed_length_bytes_exact(-128, 1),
451            Some(vec![0x80])
452        );
453        assert_eq!(decimal_to_fixed_length_bytes_exact(-129, 1), None);
454
455        assert_eq!(decimal_to_fixed_length_bytes_exact(i128::MAX, 15), None);
456        assert_eq!(decimal_to_fixed_length_bytes_exact(0, 0), None);
457        assert_eq!(decimal_to_fixed_length_bytes_exact(0, 17), None);
458    }
459}